{"as_of":"2026-08-10T17:44:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:c6da3a15c32ff3841c08b26827147650d3a2cac867278c653a7e3ab65fea595b","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":17,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":17,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":17,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":17,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-09T19:21:22.348314Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-19T20:28:39.292915Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2305.11175","last_updated":"2023-05-25T15:02:07Z","snapshot_observed_at":"2026-07-06T15:29:20.405296Z","submitted_at":"2023-05-18T17:59:42Z","title":"VisionLLM: Large Language Model is also an Open-Ended Decoder for Vision-Centric Tasks","version":2},"cited_work":{"arxiv_id":"2305.11175","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11175","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Visionllm: Large language model is also an open-ended decoder for vision-centric tasks","venue":null,"work_id":"61bd7118-a781-4cc5-bb50-fc0b2589e412","year":2023},"citing_paper":{"arxiv_id":"2306.13394","last_updated":"2025-10-24T02:45:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-23T09:22:36Z","title":"MME: A Comprehensive Evaluation Benchmark for Multimodal Large Language Models","version":5},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-10T20:25:33.854923Z"},"links":{"cited_paper":"/paper/2305.11175","citing_paper":"/paper/2306.13394"},"observation_digest":"sha256:bed9cf5231383fbeab617fb2e35eb14191dd11ea6695a2e38dd43d2ab0114ecc","observation_id":"62b9628f-a625-40ff-917a-36cafc81723f","resolution":{"observed_at":"2026-05-10T20:25:34.186308Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11175","last_updated":"2023-05-25T15:02:07Z","snapshot_observed_at":"2026-07-06T15:29:20.405296Z","submitted_at":"2023-05-18T17:59:42Z","title":"VisionLLM: Large Language Model is also an Open-Ended Decoder for Vision-Centric Tasks","version":2},"cited_work":{"arxiv_id":"2305.11175","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11175","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Visionllm: Large language model is also an open-ended decoder for vision-centric tasks","venue":null,"work_id":"61bd7118-a781-4cc5-bb50-fc0b2589e412","year":2023},"citing_paper":{"arxiv_id":"2306.13549","last_updated":"2024-11-29T15:51:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-23T15:21:52Z","title":"A Survey on Multimodal Large Language Models","version":4},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-05-16T02:56:41.658658Z"},"links":{"cited_paper":"/paper/2305.11175","citing_paper":"/paper/2306.13549"},"observation_digest":"sha256:d83facce6d8fe3ddb97a97312e83b8187f7c65e7f34f988d3952c9f78e39017d","observation_id":"43c07343-5c38-44a3-a53b-3a07a7627ca7","resolution":{"observed_at":"2026-05-16T02:56:42.616141Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11175","last_updated":"2023-05-25T15:02:07Z","snapshot_observed_at":"2026-07-06T15:29:20.405296Z","submitted_at":"2023-05-18T17:59:42Z","title":"VisionLLM: Large Language Model is also an Open-Ended Decoder for Vision-Centric Tasks","version":2},"cited_work":{"arxiv_id":"2305.11175","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11175","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Visionllm: Large language model is also an open-ended decoder for vision-centric tasks","venue":null,"work_id":"61bd7118-a781-4cc5-bb50-fc0b2589e412","year":2023},"citing_paper":{"arxiv_id":"2306.14824","last_updated":"2023-07-13T05:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-26T16:32:47Z","title":"Kosmos-2: Grounding Multimodal Large Language Models to the World","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-12T05:19:47.907355Z"},"links":{"cited_paper":"/paper/2305.11175","citing_paper":"/paper/2306.14824"},"observation_digest":"sha256:94edf2564a818f11a9c294fc5e7eb6df52eef0af5957091614e19dd21493222a","observation_id":"6365ab3c-6541-497a-8c00-108880ec5a20","resolution":{"observed_at":"2026-05-12T05:19:48.098848Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11175","last_updated":"2023-05-25T15:02:07Z","snapshot_observed_at":"2026-07-06T15:29:20.405296Z","submitted_at":"2023-05-18T17:59:42Z","title":"VisionLLM: Large Language Model is also an Open-Ended Decoder for Vision-Centric Tasks","version":2},"cited_work":{"arxiv_id":"2305.11175","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11175","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Visionllm: Large language model is also an open-ended decoder for vision-centric tasks","venue":null,"work_id":"61bd7118-a781-4cc5-bb50-fc0b2589e412","year":2023},"citing_paper":{"arxiv_id":"2307.06435","last_updated":"2024-10-17T01:10:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-12T20:01:52Z","title":"A Comprehensive Overview of Large Language Models","version":10},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-19T20:28:38.900026Z"},"links":{"cited_paper":"/paper/2305.11175","citing_paper":"/paper/2307.06435"},"observation_digest":"sha256:005a2b58a99537f0dabd93091ba93f444164ba39279b82490d73bfea450d268b","observation_id":"5bad6859-bc52-4026-8878-9eb53afb1eca","resolution":{"observed_at":"2026-05-19T20:28:39.294606Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11175","last_updated":"2023-05-25T15:02:07Z","snapshot_observed_at":"2026-07-06T15:29:20.405296Z","submitted_at":"2023-05-18T17:59:42Z","title":"VisionLLM: Large Language Model is also an Open-Ended Decoder for Vision-Centric Tasks","version":2},"cited_work":{"arxiv_id":"2305.11175","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11175","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Visionllm: Large language model is also an open-ended decoder for vision-centric tasks","venue":null,"work_id":"61bd7118-a781-4cc5-bb50-fc0b2589e412","year":2023},"citing_paper":{"arxiv_id":"2310.01415","last_updated":"2023-12-05T05:26:29Z","snapshot_observed_at":"2026-07-06T16:26:38.284922Z","submitted_at":"2023-10-02T17:59:57Z","title":"GPT-Driver: Learning to Drive with GPT","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-15T15:05:31.928650Z"},"links":{"cited_paper":"/paper/2305.11175","citing_paper":"/paper/2310.01415"},"observation_digest":"sha256:5b4781a8d7eddfadd62147da2ee7ccd31e4d36e6f7184fba8278e54d812501c6","observation_id":"bccf4ac7-ca5c-4fd2-91d4-5bcd947b58f3","resolution":{"observed_at":"2026-05-15T15:05:32.007128Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11175","last_updated":"2023-05-25T15:02:07Z","snapshot_observed_at":"2026-07-06T15:29:20.405296Z","submitted_at":"2023-05-18T17:59:42Z","title":"VisionLLM: Large Language Model is also an Open-Ended Decoder for Vision-Centric Tasks","version":2},"cited_work":{"arxiv_id":"2305.11175","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11175","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Visionllm: Large language model is also an open-ended decoder for vision-centric tasks","venue":null,"work_id":"61bd7118-a781-4cc5-bb50-fc0b2589e412","year":2023},"citing_paper":{"arxiv_id":"2310.03744","last_updated":"2024-05-15T19:22:44Z","snapshot_observed_at":"2026-07-06T16:28:22.350574Z","submitted_at":"2023-10-05T17:59:56Z","title":"Improved Baselines with Visual Instruction Tuning","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-12T19:11:33.783746Z"},"links":{"cited_paper":"/paper/2305.11175","citing_paper":"/paper/2310.03744"},"observation_digest":"sha256:77a7a0a32ad0c2f50d21872fab7c370e63c88fe1a89cc6d538a598f94ab4dd77","observation_id":"4d0830bd-85b7-4e67-8aa2-bb964dc523d4","resolution":{"observed_at":"2026-05-12T19:11:33.951814Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11175","last_updated":"2023-05-25T15:02:07Z","snapshot_observed_at":"2026-07-06T15:29:20.405296Z","submitted_at":"2023-05-18T17:59:42Z","title":"VisionLLM: Large Language Model is also an Open-Ended Decoder for Vision-Centric Tasks","version":2},"cited_work":{"arxiv_id":"2305.11175","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11175","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Visionllm: Large language model is also an open-ended decoder for vision-centric tasks","venue":null,"work_id":"61bd7118-a781-4cc5-bb50-fc0b2589e412","year":2023},"citing_paper":{"arxiv_id":"2310.09478","last_updated":"2023-11-07T18:25:48Z","snapshot_observed_at":"2026-08-06T12:38:03.720232Z","submitted_at":"2023-10-14T03:22:07Z","title":"MiniGPT-v2: large language model as a unified interface for vision-language multi-task learning","version":3},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-16T07:13:08.867745Z"},"links":{"cited_paper":"/paper/2305.11175","citing_paper":"/paper/2310.09478"},"observation_digest":"sha256:1e7f1da18ae13bdda799a700ecd1cb2ad86b0eac6f25cb7a78b08346d9838d5f","observation_id":"bb3987c3-7670-4e50-a8b8-b95fa82de8b8","resolution":{"observed_at":"2026-05-16T07:13:08.994163Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11175","last_updated":"2023-05-25T15:02:07Z","snapshot_observed_at":"2026-07-06T15:29:20.405296Z","submitted_at":"2023-05-18T17:59:42Z","title":"VisionLLM: Large Language Model is also an Open-Ended Decoder for Vision-Centric Tasks","version":2},"cited_work":{"arxiv_id":"2305.11175","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11175","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Visionllm: Large language model is also an open-ended decoder for vision-centric tasks","venue":null,"work_id":"61bd7118-a781-4cc5-bb50-fc0b2589e412","year":2023},"citing_paper":{"arxiv_id":"2311.07575","last_updated":"2023-11-13T18:59:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-11-13T18:59:47Z","title":"SPHINX: The Joint Mixing of Weights, Tasks, and Visual Embeddings for Multi-modal Large Language Models","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-17T03:03:26.723464Z"},"links":{"cited_paper":"/paper/2305.11175","citing_paper":"/paper/2311.07575"},"observation_digest":"sha256:2749b9255b2e5d487735a94498d004e6d9c7614054f8de3d062a32c4c1e64b15","observation_id":"8dd69875-73c2-416d-9104-4a03f7e2548c","resolution":{"observed_at":"2026-05-17T03:03:27.007139Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11175","last_updated":"2023-05-25T15:02:07Z","snapshot_observed_at":"2026-07-06T15:29:20.405296Z","submitted_at":"2023-05-18T17:59:42Z","title":"VisionLLM: Large Language Model is also an Open-Ended Decoder for Vision-Centric Tasks","version":2},"cited_work":{"arxiv_id":"2305.11175","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11175","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Visionllm: Large language model is also an open-ended decoder for vision-centric tasks","venue":null,"work_id":"61bd7118-a781-4cc5-bb50-fc0b2589e412","year":2023},"citing_paper":{"arxiv_id":"2401.10935","last_updated":"2024-02-23T04:36:51Z","snapshot_observed_at":"2026-08-08T03:28:12.161766Z","submitted_at":"2024-01-17T08:10:35Z","title":"SeeClick: Harnessing GUI Grounding for Advanced Visual GUI Agents","version":2},"reference_index":99,"source":"arxiv_source","source_observed_at":"2026-05-17T10:09:46.447508Z"},"links":{"cited_paper":"/paper/2305.11175","citing_paper":"/paper/2401.10935"},"observation_digest":"sha256:093e9545dec2ac517b4765d9145979e7e37a6d1773171ed97d2f271428875044","observation_id":"5e7d13db-324a-4214-862d-97aac0223663","resolution":{"observed_at":"2026-05-17T10:09:46.610091Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11175","last_updated":"2023-05-25T15:02:07Z","snapshot_observed_at":"2026-07-06T15:29:20.405296Z","submitted_at":"2023-05-18T17:59:42Z","title":"VisionLLM: Large Language Model is also an Open-Ended Decoder for Vision-Centric Tasks","version":2},"cited_work":{"arxiv_id":"2305.11175","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11175","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Visionllm: Large language model is also an open-ended decoder for vision-centric tasks","venue":null,"work_id":"61bd7118-a781-4cc5-bb50-fc0b2589e412","year":2023},"citing_paper":{"arxiv_id":"2401.15947","last_updated":"2024-12-23T08:05:14Z","snapshot_observed_at":"2026-08-06T02:31:58.372974Z","submitted_at":"2024-01-29T08:13:40Z","title":"MoE-LLaVA: Mixture of Experts for Large Vision-Language Models","version":5},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-16T02:33:30.143907Z"},"links":{"cited_paper":"/paper/2305.11175","citing_paper":"/paper/2401.15947"},"observation_digest":"sha256:7c646a5f01d4a182e2ae7d1e041f72bf9b4b34a2b5bfbd2c644ae9b44cc8b761","observation_id":"33090c72-7ad7-465e-9114-fb44dccbe226","resolution":{"observed_at":"2026-05-16T02:33:30.420855Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11175","last_updated":"2023-05-25T15:02:07Z","snapshot_observed_at":"2026-07-06T15:29:20.405296Z","submitted_at":"2023-05-18T17:59:42Z","title":"VisionLLM: Large Language Model is also an Open-Ended Decoder for Vision-Centric Tasks","version":2},"cited_work":{"arxiv_id":"2305.11175","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11175","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Visionllm: Large language model is also an open-ended decoder for vision-centric tasks","venue":null,"work_id":"61bd7118-a781-4cc5-bb50-fc0b2589e412","year":2023},"citing_paper":{"arxiv_id":"2403.09611","last_updated":"2024-04-18T18:51:04Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-03-14T17:51:32Z","title":"MM1: Methods, Analysis & Insights from Multimodal LLM Pre-training","version":4},"reference_index":115,"source":"pdf_text","source_observed_at":"2026-05-16T04:09:36.019146Z"},"links":{"cited_paper":"/paper/2305.11175","citing_paper":"/paper/2403.09611"},"observation_digest":"sha256:b6c3e259c190a6c41f45e4cedd82550038048bcb120cf0e8d703f75a4b5347bb","observation_id":"b4746f15-188b-4783-9fbd-677e6ac04446","resolution":{"observed_at":"2026-05-16T04:09:36.310691Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11175","last_updated":"2023-05-25T15:02:07Z","snapshot_observed_at":"2026-07-06T15:29:20.405296Z","submitted_at":"2023-05-18T17:59:42Z","title":"VisionLLM: Large Language Model is also an Open-Ended Decoder for Vision-Centric Tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11175","snapshot_observed_at":"2026-08-09T19:21:22.348314Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.00372","last_updated":"2025-08-14T12:58:31Z","snapshot_observed_at":"2026-08-09T19:12:46.915574Z","submitted_at":"2025-02-01T09:19:08Z","title":"NAVER: A Neuro-Symbolic Compositional Automaton for Visual Grounding with Explicit Logic Reasoning","version":3},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-09T19:21:22.348314Z"},"links":{"cited_paper":"/paper/2305.11175","citing_paper":"/paper/2502.00372"},"observation_digest":"sha256:f2566521d016fd936e914928f5c535ac91c078fad9d0f47764983824f8a1d336","observation_id":"6aaae9ad-0aa3-4802-992e-9e3a7360e4e9","resolution":{"observed_at":"2026-08-09T19:21:22.348314Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11175","last_updated":"2023-05-25T15:02:07Z","snapshot_observed_at":"2026-07-06T15:29:20.405296Z","submitted_at":"2023-05-18T17:59:42Z","title":"VisionLLM: Large Language Model is also an Open-Ended Decoder for Vision-Centric Tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11175","snapshot_observed_at":"2026-08-09T14:50:57.945485Z","title":"Wang, W., Chen, Z., Chen, X., Wu, J., Zhu, X., Zeng, G., Luo, P., Lu, T., Zhou, J., Qiao, Y ., and Dai, J","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.01719","last_updated":"2025-02-07T03:54:34Z","snapshot_observed_at":"2026-08-10T02:03:37.501894Z","submitted_at":"2025-02-03T18:56:33Z","title":"MJ-VIDEO: Fine-Grained Benchmarking and Rewarding Video Preferences in Video Generation","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-09T14:50:57.945485Z"},"links":{"cited_paper":"/paper/2305.11175","citing_paper":"/paper/2502.01719"},"observation_digest":"sha256:806942e5728d3bf467e7b358ff90913bb08a891c4aeec5d79a826e545f6cadb6","observation_id":"d820e77e-2870-4205-b739-4e60b49df5cc","resolution":{"observed_at":"2026-08-09T14:50:57.945485Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11175","last_updated":"2023-05-25T15:02:07Z","snapshot_observed_at":"2026-07-06T15:29:20.405296Z","submitted_at":"2023-05-18T17:59:42Z","title":"VisionLLM: Large Language Model is also an Open-Ended Decoder for Vision-Centric Tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11175","snapshot_observed_at":"2026-08-02T23:23:34.980496Z","title":"K., Singhal, S., Som, S., et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.14134","last_updated":"2026-06-01T07:44:02Z","snapshot_observed_at":"2026-08-09T15:21:28.970868Z","submitted_at":"2026-02-15T13:12:28Z","title":"DenseMLLM: Standard Multimodal LLMs for Dense Prediction","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-02T23:23:34.980496Z"},"links":{"cited_paper":"/paper/2305.11175","citing_paper":"/paper/2602.14134"},"observation_digest":"sha256:264d814ed2bd7b0cb568e02667e3b3bfdeac20256a579acaf1d0411b0750ad70","observation_id":"bd6dfdfe-4afe-437f-a1a5-f4490fc688d6","resolution":{"observed_at":"2026-08-02T23:23:34.980496Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11175","last_updated":"2023-05-25T15:02:07Z","snapshot_observed_at":"2026-07-06T15:29:20.405296Z","submitted_at":"2023-05-18T17:59:42Z","title":"VisionLLM: Large Language Model is also an Open-Ended Decoder for Vision-Centric Tasks","version":2},"cited_work":{"arxiv_id":"2305.11175","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11175","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Visionllm: Large language model is also an open-ended decoder for vision-centric tasks","venue":null,"work_id":"61bd7118-a781-4cc5-bb50-fc0b2589e412","year":2023},"citing_paper":{"arxiv_id":"2604.03231","last_updated":"2026-04-03T17:59:51Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-03T17:59:51Z","title":"CoME-VL: Scaling Complementary Multi-Encoder Vision-Language Learning","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-13T20:28:30.864143Z"},"links":{"cited_paper":"/paper/2305.11175","citing_paper":"/paper/2604.03231"},"observation_digest":"sha256:0c8e1b2e5933a48ec85d26d10fe6b96dd18820b887f8996785a519b17ab79720","observation_id":"6c44dd11-9231-4310-9ba9-8b2573aa5f5d","resolution":{"observed_at":"2026-05-13T20:33:17.122294Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11175","last_updated":"2023-05-25T15:02:07Z","snapshot_observed_at":"2026-07-06T15:29:20.405296Z","submitted_at":"2023-05-18T17:59:42Z","title":"VisionLLM: Large Language Model is also an Open-Ended Decoder for Vision-Centric Tasks","version":2},"cited_work":{"arxiv_id":"2305.11175","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11175","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Visionllm: Large language model is also an open-ended decoder for vision-centric tasks","venue":null,"work_id":"61bd7118-a781-4cc5-bb50-fc0b2589e412","year":2023},"citing_paper":{"arxiv_id":"2604.11789","last_updated":"2026-04-20T14:38:53Z","snapshot_observed_at":"2026-08-09T05:10:13.009841Z","submitted_at":"2026-04-13T17:55:02Z","title":"LMMs Meet Object-Centric Vision: Understanding, Segmentation, Editing and Generation","version":2},"reference_index":173,"source":"pdf_text","source_observed_at":"2026-05-10T15:35:37.095627Z"},"links":{"cited_paper":"/paper/2305.11175","citing_paper":"/paper/2604.11789"},"observation_digest":"sha256:0b0c0e212eba1be4bf2ae89e12e866c493287da96264a091aefc0f50bd5ac0a2","observation_id":"c28589f4-d1d8-47ac-a7a6-d8f8498eefe8","resolution":{"observed_at":"2026-05-11T10:11:09.601368Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11175","last_updated":"2023-05-25T15:02:07Z","snapshot_observed_at":"2026-07-06T15:29:20.405296Z","submitted_at":"2023-05-18T17:59:42Z","title":"VisionLLM: Large Language Model is also an Open-Ended Decoder for Vision-Centric Tasks","version":2},"cited_work":{"arxiv_id":"2305.11175","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11175","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Visionllm: Large language model is also an open-ended decoder for vision-centric tasks","venue":null,"work_id":"61bd7118-a781-4cc5-bb50-fc0b2589e412","year":2023},"citing_paper":{"arxiv_id":"2604.14373","last_updated":"2026-04-17T02:00:12Z","snapshot_observed_at":"2026-07-06T23:02:09.426599Z","submitted_at":"2026-04-15T19:43:20Z","title":"SatBLIP: Context Understanding and Feature Identification from Satellite Imagery with Vision-Language Learning","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-10T13:08:36.004520Z"},"links":{"cited_paper":"/paper/2305.11175","citing_paper":"/paper/2604.14373"},"observation_digest":"sha256:2f82f4fe6e49b2c0d7195a9f63c26c77f893f368bcd3ea16755f5e715ed86bf2","observation_id":"293e4431-9be4-413f-97d4-4c252a11b324","resolution":{"observed_at":"2026-05-10T13:10:26.578809Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2305.11175/citation-record","integrity":"/paper/2305.11175/integrity","json":"/paper/2305.11175/citation-record.json","paper":"/paper/2305.11175"},"outbound":[],"paper":{"arxiv_id":"2305.11175","last_updated":"2023-05-25T15:02:07Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T15:29:20.405296Z","submitted_at":"2023-05-18T17:59:42Z","title":"VisionLLM: Large Language Model is also an Open-Ended Decoder for Vision-Centric Tasks"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 17 inbound Pith citation observations for arXiv:2305.11175."}