{"as_of":"2026-08-10T16:00:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:082f7644eaa324187e8e573970bbd88e22175f56f046e63d73113da5841820ae","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":19,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":19,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":19,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":19,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T00:42:24.776729Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T20:08:55.658612Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":"2305.05662","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-07-03T20:08:55.658612Z","title":"Interngpt: Solving vision-centric tasks by interacting with chatgpt beyond lan- guage","venue":null,"work_id":"456c399b-9545-4252-b591-33d7b091fbf0","year":2023},"citing_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:00.457974Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2305.06355"},"observation_digest":"sha256:4fa8d3de37d796eafe6cf70432f96a01887d0ff9a8f99232e414fdc0e4149dca","observation_id":"dd337dbf-d56d-4135-a28d-da0da5248840","resolution":{"observed_at":"2026-05-13T23:30:00.598826Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":"2305.05662","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-07-03T20:08:55.658612Z","title":"Interngpt: Solving vision-centric tasks by interacting with chatgpt beyond lan- guage","venue":null,"work_id":"456c399b-9545-4252-b591-33d7b091fbf0","year":2023},"citing_paper":{"arxiv_id":"2307.06942","last_updated":"2024-01-04T05:00:34Z","snapshot_observed_at":"2026-07-06T15:53:46.393481Z","submitted_at":"2023-07-13T17:58:32Z","title":"InternVid: A Large-scale Video-Text Dataset for Multimodal Understanding and Generation","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-15T06:30:22.431538Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2307.06942"},"observation_digest":"sha256:75b4ce92d88f6ef37d9cf94514ff7ee362f11ceafb7315d7001b89dc0c6eaf4e","observation_id":"feddb060-ac08-41fd-85b3-b92b79efb313","resolution":{"observed_at":"2026-05-15T06:30:22.516674Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":"2305.05662","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-07-03T20:08:55.658612Z","title":"Interngpt: Solving vision-centric tasks by interacting with chatgpt beyond lan- guage","venue":null,"work_id":"456c399b-9545-4252-b591-33d7b091fbf0","year":2023},"citing_paper":{"arxiv_id":"2309.07864","last_updated":"2023-09-19T08:29:18Z","snapshot_observed_at":"2026-08-09T18:54:02.679174Z","submitted_at":"2023-09-14T17:12:03Z","title":"The Rise and Potential of Large Language Model Based Agents: A Survey","version":3},"reference_index":299,"source":"pdf_text","source_observed_at":"2026-05-11T10:47:44.152066Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2309.07864"},"observation_digest":"sha256:469ce26defda78f4e2a1220b945c9569cd12bc50492d7f5485f83556eca86450","observation_id":"b34065cb-728f-40bc-9c7d-ea7dd6ff454b","resolution":{"observed_at":"2026-05-11T10:47:54.533970Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":"2305.05662","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-07-03T20:08:55.658612Z","title":"Interngpt: Solving vision-centric tasks by interacting with chatgpt beyond lan- guage","venue":null,"work_id":"456c399b-9545-4252-b591-33d7b091fbf0","year":2023},"citing_paper":{"arxiv_id":"2312.14238","last_updated":"2024-01-15T15:23:55Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-21T18:59:31Z","title":"InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks","version":3},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-05-13T22:46:09.693156Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2312.14238"},"observation_digest":"sha256:8282dbe3e4ba95bc287e248169272f315d6c30924ed97adc3a95ec6d10a70036","observation_id":"ee97f41b-1822-4edc-85b2-da5fcc40a46f","resolution":{"observed_at":"2026-05-13T22:46:10.030518Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":"2305.05662","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-07-03T20:08:55.658612Z","title":"Interngpt: Solving vision-centric tasks by interacting with chatgpt beyond lan- guage","venue":null,"work_id":"456c399b-9545-4252-b591-33d7b091fbf0","year":2023},"citing_paper":{"arxiv_id":"2401.02458","last_updated":"2026-04-29T07:43:10Z","snapshot_observed_at":"2026-07-29T23:57:44.096283Z","submitted_at":"2024-01-04T08:00:32Z","title":"Data-Centric Foundation Models in Computational Healthcare: A Survey","version":3},"reference_index":181,"source":"pdf_text","source_observed_at":"2026-05-24T04:13:05.328492Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2401.02458"},"observation_digest":"sha256:33a1b01e888a024c194528e6d036619de3aad1351e72dda3d0622c2ea7b50a73","observation_id":"91f3c253-72ec-41f9-855e-271bfc0e9c18","resolution":{"observed_at":"2026-05-24T04:13:52.961508Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":"2305.05662","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-07-03T20:08:55.658612Z","title":"Interngpt: Solving vision-centric tasks by interacting with chatgpt beyond lan- guage","venue":null,"work_id":"456c399b-9545-4252-b591-33d7b091fbf0","year":2023},"citing_paper":{"arxiv_id":"2401.15947","last_updated":"2024-12-23T08:05:14Z","snapshot_observed_at":"2026-08-06T02:31:58.372974Z","submitted_at":"2024-01-29T08:13:40Z","title":"MoE-LLaVA: Mixture of Experts for Large Vision-Language Models","version":5},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-16T02:33:30.143907Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2401.15947"},"observation_digest":"sha256:47067b2a343553d6883a2e95e242e79c196f81a65918223406cc4bb3833ec67b","observation_id":"7336206c-68d3-48c4-b7a7-934d169d5c05","resolution":{"observed_at":"2026-05-16T02:33:30.353423Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":"2305.05662","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-07-03T20:08:55.658612Z","title":"Interngpt: Solving vision-centric tasks by interacting with chatgpt beyond lan- guage","venue":null,"work_id":"456c399b-9545-4252-b591-33d7b091fbf0","year":2023},"citing_paper":{"arxiv_id":"2404.16821","last_updated":"2024-04-29T20:24:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-04-25T17:59:19Z","title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","version":2},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-12T20:58:58.849040Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2404.16821"},"observation_digest":"sha256:04e3b7e7433a6a8b9e55ce9832f10ec5c56f251e1b6a4fe10cc0b7b1f2d79d77","observation_id":"7d452ab8-3858-4dea-a383-87f323cf3352","resolution":{"observed_at":"2026-05-12T20:58:59.117021Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":"2305.05662","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-07-03T20:08:55.658612Z","title":"Interngpt: Solving vision-centric tasks by interacting with chatgpt beyond lan- guage","venue":null,"work_id":"456c399b-9545-4252-b591-33d7b091fbf0","year":2023},"citing_paper":{"arxiv_id":"2411.10442","last_updated":"2025-04-07T09:09:39Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-11-15T18:59:27Z","title":"Enhancing the Reasoning Ability of Multimodal Large Language Models via Mixed Preference Optimization","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-16T09:16:17.150383Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2411.10442"},"observation_digest":"sha256:f1794e97f2b22d973ce798618d2b1634e9ea754458cf42e362b9732bb485a1f6","observation_id":"6fd60986-4138-4328-9308-0a4cbd49ab73","resolution":{"observed_at":"2026-05-16T09:16:17.302720Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-08-10T00:42:24.776729Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.18103","last_updated":"2025-01-30T03:01:01Z","snapshot_observed_at":"2026-08-10T13:52:12.675214Z","submitted_at":"2025-01-30T03:01:01Z","title":"Beyond Turn-taking: Introducing Text-based Overlap into Human-LLM Interactions","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-10T00:42:24.776729Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2501.18103"},"observation_digest":"sha256:83ce49adc0edfecf857738d383a0e0da34331dcb5e3d5aae5f40002d96d988dc","observation_id":"3f7d5a77-2fe2-4f49-b14d-f57a14295374","resolution":{"observed_at":"2026-08-10T00:42:24.776729Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-08-07T12:16:13.117762Z","title":"Interngpt: Solving vision-centric tasks by interacting with chatgpt beyond language","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.00123","last_updated":"2025-05-30T18:00:34Z","snapshot_observed_at":"2026-08-09T02:39:20.854550Z","submitted_at":"2025-05-30T18:00:34Z","title":"Visual Embodied Brain: Let Multimodal Large Language Models See, Think, and Control in Spaces","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-07T12:16:13.117762Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2506.00123"},"observation_digest":"sha256:34d5abc0916e0a2ad201aefd35bd4848471d2d4f4c979800e4a8dff562e53182","observation_id":"9e31acdc-ec99-411c-8518-ebe57bced002","resolution":{"observed_at":"2026-08-07T12:16:13.117762Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-08-07T04:43:51.207678Z","title":"Interngpt: Solving vision-centric tasks by interacting with chatgpt beyond language","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.09954","last_updated":"2025-06-11T17:23:41Z","snapshot_observed_at":"2026-08-07T21:45:10.885875Z","submitted_at":"2025-06-11T17:23:41Z","title":"Vision Generalist Model: A Survey","version":1},"reference_index":112,"source":"arxiv_source","source_observed_at":"2026-08-07T04:43:51.207678Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2506.09954"},"observation_digest":"sha256:a390079ac76e0e83dfa963b65d12257dc87f8b8f45b761ca7aeb00338d38d0e0","observation_id":"d21d2863-bf31-45e2-87fa-a91a5f5dbc6a","resolution":{"observed_at":"2026-08-07T04:43:51.207678Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-08-06T19:25:02.773169Z","title":"Interngpt: Solving vision-centric tasks by interacting with chatgpt beyond language","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.06272","last_updated":"2025-08-09T05:40:33Z","snapshot_observed_at":"2026-08-08T15:48:26.101646Z","submitted_at":"2025-07-08T07:46:26Z","title":"LIRA: Inferring Segmentation in Large Multi-modal Models with Local Interleaved Region Assistance","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T19:25:02.773169Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2507.06272"},"observation_digest":"sha256:16b36d9ebb827ca4a81cd093c98c25ec795594dc84cbf9cc42df82d2e2772ecd","observation_id":"4b577851-b868-4a87-a58d-e891afe63e52","resolution":{"observed_at":"2026-08-06T19:25:02.773169Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-08-06T15:57:02.802194Z","title":"Interngpt: Solving vision-centric tasks by interacting with chatgpt beyond lan- guage","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.14675","last_updated":"2025-07-19T16:03:34Z","snapshot_observed_at":"2026-08-09T00:51:52.133273Z","submitted_at":"2025-07-19T16:03:34Z","title":"Docopilot: Improving Multimodal Models for Document-Level Understanding","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T15:57:02.802194Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2507.14675"},"observation_digest":"sha256:76fc005875cea51da8cd1e171cbb7c970415987c60dc03cd876f592ed83b5e23","observation_id":"3c10a5ff-4100-4602-8d76-33c8e97745b3","resolution":{"observed_at":"2026-08-06T15:57:02.802194Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-08-06T00:59:52.580855Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04088","last_updated":"2025-08-07T03:52:48Z","snapshot_observed_at":"2026-08-08T12:10:28.774282Z","submitted_at":"2025-08-06T05:10:29Z","title":"GM-PRM: A Generative Multimodal Process Reward Model for Multimodal Mathematical Reasoning","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-06T00:59:52.580855Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2508.04088"},"observation_digest":"sha256:32b973ccc166b1374eecf66f8551b12354241ef602da46cf52eb3bc9b3451d21","observation_id":"2e4ab4fc-1b0e-4b94-b490-7cca333b0fa0","resolution":{"observed_at":"2026-08-06T00:59:52.580855Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":"2305.05662","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-07-03T20:08:55.658612Z","title":"Interngpt: Solving vision-centric tasks by interacting with chatgpt beyond lan- guage","venue":null,"work_id":"456c399b-9545-4252-b591-33d7b091fbf0","year":2023},"citing_paper":{"arxiv_id":"2604.03231","last_updated":"2026-04-03T17:59:51Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-03T17:59:51Z","title":"CoME-VL: Scaling Complementary Multi-Encoder Vision-Language Learning","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-13T20:28:30.864143Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2604.03231"},"observation_digest":"sha256:c3a308010ecf826ee8e9be47d88dac2280f09281abf43c42466156776e01e37b","observation_id":"2f6371ee-f53a-4db8-9243-894cc83ee150","resolution":{"observed_at":"2026-05-13T20:33:17.105582Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":"2305.05662","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-07-03T20:08:55.658612Z","title":"Interngpt: Solving vision-centric tasks by interacting with chatgpt beyond lan- guage","venue":null,"work_id":"456c399b-9545-4252-b591-33d7b091fbf0","year":2023},"citing_paper":{"arxiv_id":"2604.15670","last_updated":"2026-07-24T04:48:23Z","snapshot_observed_at":"2026-08-02T16:09:56.976755Z","submitted_at":"2026-04-17T03:48:56Z","title":"PixDLM: A Dual-Path Multimodal Language Model for UAV Reasoning Segmentation","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-10T09:20:54.635375Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2604.15670"},"observation_digest":"sha256:13bf338a5671f088943beaa9d4b1b6874b70be8c88f50042bb5efffb8b3f5296","observation_id":"64140c31-6391-473c-8d4b-e3a21d9a9c6d","resolution":{"observed_at":"2026-05-10T09:23:37.122437Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-08-02T16:10:01.893221Z","title":"Interngpt: Solving vision-centric tasks by interacting with chatgpt beyond language.arXiv preprint arXiv:2305.05662, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2604.15670","last_updated":"2026-07-24T04:48:23Z","snapshot_observed_at":"2026-08-02T16:09:56.976755Z","submitted_at":"2026-04-17T03:48:56Z","title":"PixDLM: A Dual-Path Multimodal Language Model for UAV Reasoning Segmentation","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-02T16:10:01.893221Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2604.15670"},"observation_digest":"sha256:e652cc1b117c2c09566799779edb971829ea2397c64e3c8ad5dda848df97fd58","observation_id":"6cac5e5d-667c-4c06-a6b4-08b10b62f271","resolution":{"observed_at":"2026-08-02T16:10:01.893221Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":"2305.05662","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-07-03T20:08:55.658612Z","title":"Interngpt: Solving vision-centric tasks by interacting with chatgpt beyond lan- guage","venue":null,"work_id":"456c399b-9545-4252-b591-33d7b091fbf0","year":2023},"citing_paper":{"arxiv_id":"2606.12555","last_updated":"2026-07-02T19:10:03Z","snapshot_observed_at":"2026-07-12T14:16:32.755376Z","submitted_at":"2026-06-10T18:06:27Z","title":"AudioX-Turbo: A Unified Framework for Efficient Anything-to-Audio Generation","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-27T08:04:48.283908Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2606.12555"},"observation_digest":"sha256:44c93dccb645a171334d8ed7d4e35976f601ede9fbefb05d1354bb982e41d91e","observation_id":"1529a806-b412-4f95-b41e-e8133415e65a","resolution":{"observed_at":"2026-07-03T13:28:18.704967Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language","version":4},"cited_work":{"arxiv_id":"2305.05662","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.05662","snapshot_observed_at":"2026-07-03T20:08:55.658612Z","title":"Interngpt: Solving vision-centric tasks by interacting with chatgpt beyond lan- guage","venue":null,"work_id":"456c399b-9545-4252-b591-33d7b091fbf0","year":2023},"citing_paper":{"arxiv_id":"2606.17539","last_updated":"2026-06-16T05:32:39Z","snapshot_observed_at":"2026-07-06T23:53:07.149182Z","submitted_at":"2026-06-16T05:32:39Z","title":"Reinforcing Dual-Path Reasoning in Spatial Vision Language Models","version":1},"reference_index":102,"source":"arxiv_source","source_observed_at":"2026-06-27T01:42:30.005911Z"},"links":{"cited_paper":"/paper/2305.05662","citing_paper":"/paper/2606.17539"},"observation_digest":"sha256:c15d13e783ceac6331cbf44c5694fc4a994ade02e96ada68270d47adc55d2730","observation_id":"2724805f-f09a-4d6c-bea4-08dd5cc7b783","resolution":{"observed_at":"2026-07-03T20:08:55.660752Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2305.05662/citation-record","integrity":"/paper/2305.05662/integrity","json":"/paper/2305.05662/citation-record.json","paper":"/paper/2305.05662"},"outbound":[],"paper":{"arxiv_id":"2305.05662","last_updated":"2023-06-02T16:19:48Z","latest_version":4,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T15:25:12.782361Z","submitted_at":"2023-05-09T17:58:34Z","title":"InternGPT: Solving Vision-Centric Tasks by Interacting with ChatGPT Beyond Language"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 19 inbound Pith citation observations for arXiv:2305.05662."}