{"as_of":"2026-08-11T17:13:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:dcdb43df56b31f0936fce48d6c3c8257d3011548c9cea59374ce157039b73c9d","coverage":[{"denominator":44,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":44,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T05:28:31.007992Z","state":"measured"},{"denominator":44,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":44,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2508.01805/citation-record","integrity":"/paper/2508.01805/integrity","json":"/paper/2508.01805/citation-record.json","paper":"/paper/2508.01805"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:28:37.771893Z","title":"Attention is all you need,","venue":null,"work_id":"0f1639ce-ad8c-4775-b55a-63ca9c8eb0df","year":2017},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:27.162115Z"},"links":{"citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:541aafc9d3a2bac916bb7c3f6666468c8c985c3099688a4caaf35d1c2653221c","observation_id":"58c4b7eb-5136-45ed-9d07-f7ef1e6eed79","resolution":{"observed_at":"2026-08-06T05:28:37.887978Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:28:37.608174Z","title":"Flamingo: a visual language model for few-shot learning,","venue":null,"work_id":"3b50f3db-ccf1-4859-9f4b-f97e20e0dbad","year":2022},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:27.244021Z"},"links":{"citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:6d905ea4a0d72dbc4bb5252cef48706b6d0db1df324f10c32050896b20dc3fd9","observation_id":"ed321ea6-c919-4db1-908b-82444a3d586c","resolution":{"observed_at":"2026-08-06T05:28:37.678178Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:28:37.313756Z","title":"BLIP-2: Bootstrapping language- image pre-training with frozen image encoders and large language models,","venue":null,"work_id":"e93f7f0b-90f0-4572-a225-0f1d47f6dbb4","year":2023},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:27.357176Z"},"links":{"citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:b1c503be177d44d0a9e663bdc52e08e2753879a93308fd0425614eb803709c21","observation_id":"b15e45c5-ffd7-4bc5-a1bc-8c6eecc18dd3","resolution":{"observed_at":"2026-08-06T05:28:37.468973Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:28:37.034321Z","title":"Visual instruction tuning,","venue":null,"work_id":"79825ab6-1352-496e-a45a-84e681c68728","year":2023},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:27.480107Z"},"links":{"citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:77828772951e0e958e559c79238bbbe0ddf78e1880d1bf2d148d0887fc1ba01d","observation_id":"77a02c35-d8e1-46ea-aab5-a0461c5db666","resolution":{"observed_at":"2026-08-06T05:28:37.154463Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:28:36.804074Z","title":"GPT-4V(ision) system card,","venue":null,"work_id":"c6069be5-de5b-481c-9b64-d58d63e27553","year":2023},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:27.566506Z"},"links":{"citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:3ddc251edd9bc9991e176533d1d49c7f9b06a05dc31a5f5e1e69158380f3197e","observation_id":"d7c8ea46-75e0-40fd-a5c4-67bee9205e83","resolution":{"observed_at":"2026-08-06T05:28:36.901875Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-06T05:28:27.682913Z","title":"Gemini: A family of highly capable multimodal models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:27.682913Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:61875d367992f80d3695d9e2d0413937d387cc572987401f7e5ff3b1b0764e2a","observation_id":"a7de76dd-a279-4fba-bad9-47a991aed3c7","resolution":{"observed_at":"2026-08-06T05:28:27.682913Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.13165","last_updated":"2023-11-22T05:15:12Z","snapshot_observed_at":"2026-08-10T23:50:37.862796Z","submitted_at":"2023-11-22T05:15:12Z","title":"Multimodal Large Language Models: A Survey","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.13165","snapshot_observed_at":"2026-08-06T05:28:27.791343Z","title":"Multimodal Large Language Models: A Survey,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:27.791343Z"},"links":{"cited_paper":"/paper/2311.13165","citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:4e5e47ef94ea2540b81ac32191a82b23d10b3155f77e5cf719cd1b79725d49c7","observation_id":"3a6a6579-913a-4db8-8c3f-40291285f713","resolution":{"observed_at":"2026-08-06T05:28:27.791343Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:28:36.519797Z","title":"Learning transferable visual models from natural language supervision,","venue":null,"work_id":"06b43905-df7c-4e25-b266-e52fb0708cda","year":2021},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:27.907772Z"},"links":{"citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:59e8ed14996ddb2fa6cd517754fe93044320a1081dc07de7fe97b7e355af5173","observation_id":"4b7ad9f2-723d-495f-8c70-bbbcceb3df7e","resolution":{"observed_at":"2026-08-06T05:28:36.663757Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:28:36.281518Z","title":"GPipe: Efficient training of giant neural networks using pipeline parallelism,","venue":null,"work_id":"6184156e-c381-42b3-9b97-db956a526676","year":2019},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:28.008715Z"},"links":{"citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:318c16e1d8c13a382fc2233d9828cea00df221661e7acb2e53d2a82dd615d313","observation_id":"1c579584-1d74-401b-bf4d-c766e6744607","resolution":{"observed_at":"2026-08-06T05:28:36.382048Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:28:36.103471Z","title":"Are we ready for autonomous driving? The KITTI vision benchmark suite,","venue":null,"work_id":"a2880b5e-3790-4589-8d01-5f4f69c0a860","year":2012},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:28.112758Z"},"links":{"citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:2e2a813000116757cf9bb4a599bc3a7097f4788f4d6363263b787508b78d75d6","observation_id":"2fc2a36b-fc36-4aed-bc34-11bb0d7e18c6","resolution":{"observed_at":"2026-08-06T05:28:36.161539Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:28:35.861328Z","title":"CheXpert: A large chest radiograph dataset with uncertainty labels and expert comparison,","venue":null,"work_id":"eace71c8-0a59-4aee-8799-e648924dd732","year":2019},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:28.245719Z"},"links":{"citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:2ffcd428a34ef79f91bb54ef097f377c69024cbc512cde5dd6348c10d0101456","observation_id":"e968f2ef-eb65-473b-a658-f297bfcc937b","resolution":{"observed_at":"2026-08-06T05:28:35.979321Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2108.07258","last_updated":"2022-07-12T23:45:14Z","snapshot_observed_at":"2026-08-02T09:20:40.804790Z","submitted_at":"2021-08-16T17:50:08Z","title":"On the Opportunities and Risks of Foundation Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2108.07258","snapshot_observed_at":"2026-08-06T05:28:28.330922Z","title":"On the opportunities and risks of foundation models,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:28.330922Z"},"links":{"cited_paper":"/paper/2108.07258","citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:69cd746873d9aa0e167d98d9899b5d9b5169cd77407ca80e5017b8489af58cd7","observation_id":"4693fc8e-b960-4e65-a0da-3112d475814f","resolution":{"observed_at":"2026-08-06T05:28:28.330922Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:28:35.611999Z","title":"Outrageously large neural networks: The sparsely-gated mixture-of-experts layer,","venue":null,"work_id":"c0783660-d3d4-414a-b3aa-ed84d35c672f","year":2017},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:28.464722Z"},"links":{"citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:d4e51fad4faf3c1774b6be0d41fc8799acb5d360e20cf06c5b936a4731eab739","observation_id":"ded83ce8-1ede-430b-8f3a-0071dfc61015","resolution":{"observed_at":"2026-08-06T05:28:35.696692Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.13046","last_updated":"2024-10-31T17:39:34Z","snapshot_observed_at":"2026-08-05T02:05:40.291312Z","submitted_at":"2024-04-19T17:59:48Z","title":"MoVA: Adapting Mixture of Vision Experts to Multimodal Context","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.13046","snapshot_observed_at":"2026-08-06T05:28:28.582750Z","title":"MoV A: Adapting mixture of vision experts to multimodal context,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:28.582750Z"},"links":{"cited_paper":"/paper/2404.13046","citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:9b41a80bbd9ca7de98896b390c1e49c7def87ce6565636769cef5e02ba965a81","observation_id":"6dbc7e4e-478f-414e-ac8f-4d80661a55fa","resolution":{"observed_at":"2026-08-06T05:28:28.582750Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.15947","last_updated":"2024-12-23T08:05:14Z","snapshot_observed_at":"2026-08-06T02:31:58.372974Z","submitted_at":"2024-01-29T08:13:40Z","title":"MoE-LLaVA: Mixture of Experts for Large Vision-Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.15947","snapshot_observed_at":"2026-08-06T05:28:28.723551Z","title":"MoE-LLaV A: Mixture of experts for large vision-language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:28.723551Z"},"links":{"cited_paper":"/paper/2401.15947","citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:ef87ae6085381e903164b4c41d2a709b06f5f561d67a7fa1713a5ef475ccc161","observation_id":"fc9e6c91-c1ad-4e8a-853f-ccccfda0095e","resolution":{"observed_at":"2026-08-06T05:28:28.723551Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.11273","last_updated":"2024-05-18T12:16:01Z","snapshot_observed_at":"2026-08-03T11:46:26.153374Z","submitted_at":"2024-05-18T12:16:01Z","title":"Uni-MoE: Scaling Unified Multimodal LLMs with Mixture of Experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.11273","snapshot_observed_at":"2026-08-06T05:28:28.885000Z","title":"Uni-MoE: Scaling Unified Multimodal LLMs with Mixture of Experts,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:28.885000Z"},"links":{"cited_paper":"/paper/2405.11273","citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:a9d3011cd0ad3be32513b7b84b8c6ec2af52763d363eb5f00cf4edf460ee264b","observation_id":"800b636e-a3e2-47da-b929-62c22976b28c","resolution":{"observed_at":"2026-08-06T05:28:28.885000Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07575","last_updated":"2023-11-13T18:59:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-11-13T18:59:47Z","title":"SPHINX: The Joint Mixing of Weights, Tasks, and Visual Embeddings for Multi-modal Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.07575","snapshot_observed_at":"2026-08-06T05:28:29.028589Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:29.028589Z"},"links":{"cited_paper":"/paper/2311.07575","citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:4056bbcf77c3bacdf00b1a655861c547cc9a23b39102d04d6953929d6012a518","observation_id":"5053dd87-4df4-425f-a960-ea5c9efa6517","resolution":{"observed_at":"2026-08-06T05:28:29.028589Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:28:35.401499Z","title":"Quantization and training of neural networks for efficient integer-arithmetic-only inference,","venue":null,"work_id":"79511eeb-c716-4ee8-8551-957e73dcf11e","year":2018},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:29.118082Z"},"links":{"citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:88a478bd49ddbf6870d4c10353822f450f102a1864c51d08bdf923677626ef64","observation_id":"a1c95fab-a0bd-4c02-bdee-1baf6f6db446","resolution":{"observed_at":"2026-08-06T05:28:35.498999Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:28:35.154455Z","title":"Semantic communi- cations for future Internet: Fundamentals, applications, and challenges,","venue":null,"work_id":"140194fb-13cd-48c2-aabc-03a01b49de8c","year":2023},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:29.175454Z"},"links":{"citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:a2cd75441557107ca42c0bd60c9a5d29c160e6f8b40c25496cc34f3b3a5480a2","observation_id":"7f590711-5ed2-4868-9ad3-f1c73bf8941a","resolution":{"observed_at":"2026-08-06T05:28:35.259834Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:28:34.890498Z","title":"Model Context Protocol: An open standard for connecting AI assistants to the world,","venue":null,"work_id":"bf0e1fff-47e8-46cd-9107-e487fd068a18","year":2024},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:29.274728Z"},"links":{"citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:2947a0193003a72a9f51f001d6c6be76cd032fbb043784f485acab79cf83e0bb","observation_id":"7d2684f1-2d5e-4d65-bbb2-20e441305abc","resolution":{"observed_at":"2026-08-06T05:28:35.048642Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:28:34.665510Z","title":"Soft actor-critic: Off- policy maximum entropy deep reinforcement learning with a stochastic actor,","venue":null,"work_id":"4a146d9e-c680-4712-b69f-66a7306d1dbc","year":2018},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:29.371945Z"},"links":{"citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:a4abb2148fa616423e0198e1f679cca54be2a0033918d3e54946c08e425a0253","observation_id":"2a38a444-a312-4dcd-892d-59ae20333ce1","resolution":{"observed_at":"2026-08-06T05:28:34.758087Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:28:34.403085Z","title":"Variational inference: A review for statisticians,","venue":null,"work_id":"e8cabd0f-a4a2-4346-ac6d-321e9688368c","year":2017},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:29.453585Z"},"links":{"citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:3f21ab80e15ba73289a65834fbb7802c3d396bef3ee4cf4f7d9787dcd283776f","observation_id":"cce9d416-f28e-4688-b77b-f1004b543d4d","resolution":{"observed_at":"2026-08-06T05:28:34.538165Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:28:34.151400Z","title":null,"venue":null,"work_id":"a896408d-0a47-4ca2-999a-121cd2c8c5b6","year":2002},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:29.523900Z"},"links":{"citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:0aceedf2d24d9dca4601ab949aaa5237c9f339f92ea8d638ee1cf7fd3d62f0d2","observation_id":"91906a3f-c0c1-4c59-989a-69f55e9b67da","resolution":{"observed_at":"2026-08-06T05:28:34.258420Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:28:29.582311Z","title":"Goldsmith, Wireless Communications","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:29.582311Z"},"links":{"citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:efc48f1acd4e7a2fe643d74b7e65d90b72feebcceb4d283d6399b43f2c328234","observation_id":"8d491f23-dcb5-4121-b3bb-c2effb726810","resolution":{"observed_at":"2026-08-06T05:28:29.582311Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:28:29.658115Z","title":"Correlation model for shadow fading in mobile radio systems,","venue":null,"work_id":null,"year":1991},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:29.658115Z"},"links":{"citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:84dc9128af0f33b73d5afb28fded8d8cf7beff675a417d92bbc01818a420c5d7","observation_id":"0362e2d8-cdf2-47dc-a51d-fdc6fd1c5fe3","resolution":{"observed_at":"2026-08-06T05:28:29.658115Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:28:29.723707Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:29.723707Z"},"links":{"citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:d1b63d76360df66285e50a72aa6fca92b22f8c5e0f54f785a0c858e424b00e6c","observation_id":"b0baa3cc-65b8-466e-955f-beb4bfd46632","resolution":{"observed_at":"2026-08-06T05:28:29.723707Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:28:29.784574Z","title":"Billion-scale similarity search with GPUs,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:29.784574Z"},"links":{"citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:af39c40849b139d752cd2e271417c0f0a357dbbe30bc70f6502db2c6bd9e87b7","observation_id":"36a5f0af-7e02-4208-ac75-e46934ddb293","resolution":{"observed_at":"2026-08-06T05:28:29.784574Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1186/s13638-019-1517-y","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"Design of coherence- aware channel indication and prediction for rate adaptation,","venue":"EURASIP Journal on Wireless Communications and Networking","work_id":"9b70cc42-3c35-4dd0-ad9f-f6d3b33c2803","year":2019},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:29.850481Z"},"links":{"citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:5320cbb10939bf5d8151e2938ca93068fe4a3e638fcba96fb12cf94350e5aec5","observation_id":"0da89616-29ca-4e87-a233-58bc37123e45","resolution":{"observed_at":"2026-08-06T05:28:31.716166Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.2307/1271526","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"Bayesian Forecasting and Dynamic Models,","venue":"Technometrics","work_id":"aa2b3d26-32d1-4273-a00f-51228b293aac","year":1997},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:29.908023Z"},"links":{"citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:541789f5ea2d88bb8b9b74b0e9b7471b08b4dbacb6f4d3a569fa69887be68f77","observation_id":"7ce1394c-4d72-4683-a49f-41b8d0a5dfd1","resolution":{"observed_at":"2026-08-06T05:28:31.477624Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.3390/e26110959","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"Exact Expressions for Kullback–Leibler Divergence for Univariate Distributions,","venue":"Entropy","work_id":"afc27183-503a-43f5-a1a9-3857ece92d1d","year":2024},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:29.970215Z"},"links":{"citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:f8d923994b8170f0fa523dc6d3a20252094e36d8959446106733239e460479f4","observation_id":"28e9d839-f3f2-4cc9-8531-de9553b01419","resolution":{"observed_at":"2026-08-06T05:28:31.241572Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.13394","last_updated":"2025-10-24T02:45:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-23T09:22:36Z","title":"MME: A Comprehensive Evaluation Benchmark for Multimodal Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.13394","snapshot_observed_at":"2026-08-06T05:28:30.019543Z","title":"MME: A comprehensive evaluation benchmark for multimodal large language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:30.019543Z"},"links":{"cited_paper":"/paper/2306.13394","citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:184d60dbac80b999bd504103c262e62d18641a7d3c00c4d05c7448270604747a","observation_id":"f56c7da4-9cd1-4f91-9c7f-8c997624180d","resolution":{"observed_at":"2026-08-06T05:28:30.019543Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:28:33.849253Z","title":"Learn to explain: Multimodal reasoning via thought chains for science question answering,","venue":null,"work_id":"d698dd82-4a4b-4059-9147-2816057c566e","year":2022},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:30.065261Z"},"links":{"citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:fb54a3f0c7faa6d186fb306c9071d402e01036f599d54d79b8a68ab55c4fa5b3","observation_id":"a31d8df1-eb42-4528-9056-a0e2f1567566","resolution":{"observed_at":"2026-08-06T05:28:33.966178Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:28:33.642340Z","title":"EdgeViT: Efficient visual modeling for edge computing,","venue":null,"work_id":"ae0f684a-0d97-4f94-be49-0f44419f1d9b","year":2022},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:30.150058Z"},"links":{"citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:38565bfbb32eacc3f6c59cefbd65804f4fcc82d36078930795480f355060a0b1","observation_id":"95778c97-8116-42e5-b4f8-c7e2cab0d1e7","resolution":{"observed_at":"2026-08-06T05:28:33.744255Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.07193","last_updated":"2024-02-02T10:24:09Z","snapshot_observed_at":"2026-08-11T10:12:11.384939Z","submitted_at":"2023-04-14T15:12:19Z","title":"DINOv2: Learning Robust Visual Features without Supervision","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.07193","snapshot_observed_at":"2026-08-06T05:28:30.225411Z","title":"DINOv2: Learning Robust Visual Features without Supervision,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:30.225411Z"},"links":{"cited_paper":"/paper/2304.07193","citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:c7907e1f87d5acf3ebbbadbde500d6618b8e6b28025216865c87777f6d72ef6d","observation_id":"08aff055-873e-4ef2-bb1c-568401557bfb","resolution":{"observed_at":"2026-08-06T05:28:30.225411Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:28:33.439251Z","title":"Carion, F","venue":null,"work_id":"5ad47bd7-387c-4e48-a064-e5f60a1cd872","year":2020},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:30.260805Z"},"links":{"citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:56a7f325e8f0bdd75f2f79d5b6e992ae4888330de58d25a05ff5aef1e2ee01c1","observation_id":"9058a541-53e6-438f-a111-27bfd4bec2f3","resolution":{"observed_at":"2026-08-06T05:28:33.517529Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:28:33.169086Z","title":"”Segment anything.” Proceedings of the IEEE/CVF international conference on computer vision","venue":null,"work_id":"3c7e0bf2-b347-4e39-825f-0bd01170f8d2","year":2023},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:30.311042Z"},"links":{"citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:c4603a41474a3a512972df3aa9de25842a46cc08fe3d5f0b28b88b66483755fa","observation_id":"7863eb52-77b9-423b-a111-ea6429067e00","resolution":{"observed_at":"2026-08-06T05:28:33.272765Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:28:32.991110Z","title":"”Pix2struct: Screenshot parsing as pretraining for visual language understanding.” International Conference on Machine Learning","venue":null,"work_id":"fa290a02-6b44-40c2-8d77-29fb8e77f2fc","year":2023},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:30.385137Z"},"links":{"citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:1aca2379c227298af13973d66f9433b6697eefce0f2a98eb8d2942c50b61dd1f","observation_id":"ce69559f-c102-46ab-b8e5-a147c4252005","resolution":{"observed_at":"2026-08-06T05:28:33.075370Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:28:30.440513Z","title":"DePlot: One-shot visual language reasoning by plot-to-table translation,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:30.440513Z"},"links":{"citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:97495c18894a4c8bc067f4e7aae2d92d52f8e82512531e6195b49e3b87f3890f","observation_id":"be705dd5-c18e-4b77-9ace-73c05ce9de7e","resolution":{"observed_at":"2026-08-06T05:28:30.440513Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:28:32.750723Z","title":null,"venue":null,"work_id":"29aaf821-e45d-44b4-93b6-ff901de46d8c","year":2024},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:30.521602Z"},"links":{"citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:02c79a6918c525988af6ea40a9f0d2c55ea75bad899c20aa433ec1e09713daa6","observation_id":"2746403a-70de-4f48-8073-6af1059634fe","resolution":{"observed_at":"2026-08-06T05:28:32.867902Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.00915","last_updated":"2025-01-08T22:58:51Z","snapshot_observed_at":"2026-07-06T14:57:39.647497Z","submitted_at":"2023-03-02T02:20:04Z","title":"BiomedCLIP: a multimodal biomedical foundation model pretrained from fifteen million scientific image-text pairs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.00915","snapshot_observed_at":"2026-08-06T05:28:30.630380Z","title":"Zhang, Y","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:30.630380Z"},"links":{"cited_paper":"/paper/2303.00915","citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:0504adcab035d18315d1e75e5923f6def50cfc63c27df1c8714a105dfd047201","observation_id":"50dad1d6-a554-4ca9-a40a-e9e7f5189ced","resolution":{"observed_at":"2026-08-06T05:28:30.630380Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.00445","last_updated":"2022-05-01T11:01:28Z","snapshot_observed_at":"2026-08-07T07:47:07.886786Z","submitted_at":"2022-05-01T11:01:28Z","title":"MRKL Systems: A modular, neuro-symbolic architecture that combines large language models, external knowledge sources and discrete reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.00445","snapshot_observed_at":"2026-08-06T05:28:30.709973Z","title":"MRKL systems: A modular, neuro-symbolic architecture that combines large language models, external knowledge sources and discrete reasoning,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:30.709973Z"},"links":{"cited_paper":"/paper/2205.00445","citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:b2bb2a0fe1aeeffb0cde3eaff9ea9451511db2dc019b22a9a0466267289077d6","observation_id":"6da73563-e187-441f-8a2f-5f71241f4148","resolution":{"observed_at":"2026-08-06T05:28:30.709973Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:28:32.555802Z","title":"Resource manage- ment with deep reinforcement learning,","venue":null,"work_id":"bcd57615-432e-4cdb-81a9-d1edad3c07d2","year":2016},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:30.819463Z"},"links":{"citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:82a77545f6aa18e7fc2af4636fe604ff0849a6c9da2ae228fe6bb4f71c48586b","observation_id":"c4d33446-d77e-48f7-a07a-b4a0f4f72c5a","resolution":{"observed_at":"2026-08-06T05:28:32.621476Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:28:32.416166Z","title":"Human-level control through deep reinforcement learning,","venue":null,"work_id":"b0a9dc7f-f7d5-4902-9ff7-0dad7cdaedec","year":null},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:30.918159Z"},"links":{"citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:e9eaf1f641cc529ca3aec9ab3156c67a149c90b9f07ba8cd9fc15954602aa6d6","observation_id":"c0ffb6cc-622d-4746-a4ae-1b65ca98c86e","resolution":{"observed_at":"2026-08-06T05:28:32.486135Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T05:28:32.216276Z","title":"Adaptive computation time for recurrent neural networks,","venue":null,"work_id":"bc1b0921-3cc9-41b5-b02d-9b68b09daf07","year":2016},"citing_paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T05:28:31.007992Z"},"links":{"citing_paper":"/paper/2508.01805"},"observation_digest":"sha256:ef911433b73d833fcab1d5db23a069048ce3a286da2915424c6be383a50c4aed","observation_id":"46009e0b-efe7-40a3-a11a-4b523ee682f1","resolution":{"observed_at":"2026-08-06T05:28:32.290618Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2508.01805","last_updated":"2025-08-03T15:42:05Z","latest_version":1,"primary_category":"cs.NI","snapshot_observed_at":"2026-08-10T23:50:31.211868Z","submitted_at":"2025-08-03T15:42:05Z","title":"M3LLM: Model Context Protocol-aided Mixture of Vision Experts For Multimodal LLMs in Networks"},"reference_resolution":{"displayed":44,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":18,"verified_exact":3,"verified_fuzzy":23},"total_outbound_references":44},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 44 of 44 outbound references and 0 inbound Pith citation observations for arXiv:2508.01805."}