{"as_of":"2026-08-14T11:57:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:6a379d81731ab77d02e9e32eaa60c1b3be6b214d9195677f58f46cc7813c67a0","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":42,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":42,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-14T06:32:32.682623+00:00","state":"measured"},{"denominator":42,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":42,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T21:07:09.556330Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T06:39:37.654958Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":"2311.16103","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-07-04T06:39:37.654958Z","title":"Video-bench: A com- prehensive benchmark and toolkit for evaluating video-based 10 large language models","venue":null,"work_id":"37c46a44-ccc9-4891-a106-29a0fb8d3d27","year":2023},"citing_paper":{"arxiv_id":"2306.13549","last_updated":"2024-11-29T15:51:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-23T15:21:52Z","title":"A Survey on Multimodal Large Language Models","version":4},"reference_index":133,"source":"pdf_text","source_observed_at":"2026-05-16T02:56:41.658658Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2306.13549"},"observation_digest":"sha256:f44c05a8604aac982d7009597c8a128ae9409c69e9a33166fa5d019b04801f7a","observation_id":"6a496b85-6cc0-457c-81f2-327b2ba5c964","resolution":{"observed_at":"2026-05-16T02:56:41.875900Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":"2311.16103","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-07-04T06:39:37.654958Z","title":"Video-bench: A com- prehensive benchmark and toolkit for evaluating video-based 10 large language models","venue":null,"work_id":"37c46a44-ccc9-4891-a106-29a0fb8d3d27","year":2023},"citing_paper":{"arxiv_id":"2403.00476","last_updated":"2024-06-03T04:13:39Z","snapshot_observed_at":"2026-08-12T14:24:43.915700Z","submitted_at":"2024-03-01T12:02:19Z","title":"TempCompass: Do Video LLMs Really Understand Videos?","version":3},"reference_index":110,"source":"arxiv_source","source_observed_at":"2026-05-17T02:46:16.632743Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2403.00476"},"observation_digest":"sha256:13495d2cb070df56558164819e23eb7bd7231cadfb6dba1fe7f3125621b4fecc","observation_id":"eeefd3dd-f688-43d6-80e2-93659ab49552","resolution":{"observed_at":"2026-05-17T02:46:16.779312Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":"2311.16103","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-07-04T06:39:37.654958Z","title":"Video-bench: A com- prehensive benchmark and toolkit for evaluating video-based 10 large language models","venue":null,"work_id":"37c46a44-ccc9-4891-a106-29a0fb8d3d27","year":2023},"citing_paper":{"arxiv_id":"2406.04264","last_updated":"2025-01-01T15:53:58Z","snapshot_observed_at":"2026-08-03T20:38:36.602554Z","submitted_at":"2024-06-06T17:09:32Z","title":"MLVU: Benchmarking Multi-task Long Video Understanding","version":3},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-14T19:55:26.333923Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2406.04264"},"observation_digest":"sha256:ee7e0b624639b6cbc3f94107d2548d8050581908764bc67e559d376450f6715a","observation_id":"0fe0559f-075c-4d14-a087-1de07c63ef1d","resolution":{"observed_at":"2026-05-14T19:55:26.421249Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":"2311.16103","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-07-04T06:39:37.654958Z","title":"Video-bench: A com- prehensive benchmark and toolkit for evaluating video-based 10 large language models","venue":null,"work_id":"37c46a44-ccc9-4891-a106-29a0fb8d3d27","year":2023},"citing_paper":{"arxiv_id":"2407.03320","last_updated":"2024-07-03T17:59:21Z","snapshot_observed_at":"2026-08-13T05:45:31.410659Z","submitted_at":"2024-07-03T17:59:21Z","title":"InternLM-XComposer-2.5: A Versatile Large Vision Language Model Supporting Long-Contextual Input and Output","version":1},"reference_index":111,"source":"pdf_text","source_observed_at":"2026-05-17T10:46:28.447347Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2407.03320"},"observation_digest":"sha256:88f68ad82490aa94c30ee32a245508673d8ddb3599e9a2b3b30820041fd6fd70","observation_id":"4380602a-a411-4aec-a9ef-e9139b5d7aa6","resolution":{"observed_at":"2026-05-17T10:46:28.564426Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-08-12T21:07:09.556330Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.09105","last_updated":"2025-07-01T03:47:15Z","snapshot_observed_at":"2026-08-14T05:35:02.632470Z","submitted_at":"2024-11-14T00:26:26Z","title":"VideoCogQA: A Controllable Benchmark for Evaluating Cognitive Abilities in Video-Language Models","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-12T21:07:09.556330Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2411.09105"},"observation_digest":"sha256:dfc9ec769a7d262e5b06c837ee5fc4f7b7cd2cd31f60608f30fed4679b5c6fc9","observation_id":"e5477cd8-9c92-4497-8925-47646329c903","resolution":{"observed_at":"2026-08-12T21:07:09.556330Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-08-12T17:07:00.809662Z","title":"Video-bench: A comprehen- sive benchmark and toolkit for evaluating video-based large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.12951","last_updated":"2025-03-17T02:21:38Z","snapshot_observed_at":"2026-08-12T18:26:22.658520Z","submitted_at":"2024-11-20T00:47:17Z","title":"On the Consistency of Video Large Language Models in Temporal Comprehension","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-12T17:07:00.809662Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2411.12951"},"observation_digest":"sha256:26fa95c74c5aec24ff1006f91a6098b7db5f24380d5362430191b4e2ce2fceaa","observation_id":"081f7dd8-e3a6-4ac6-a1ad-3459c8ad70bb","resolution":{"observed_at":"2026-08-12T17:07:00.809662Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":"2311.16103","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-07-04T06:39:37.654958Z","title":"Video-bench: A com- prehensive benchmark and toolkit for evaluating video-based 10 large language models","venue":null,"work_id":"37c46a44-ccc9-4891-a106-29a0fb8d3d27","year":2023},"citing_paper":{"arxiv_id":"2411.16771","last_updated":"2026-04-23T13:21:19Z","snapshot_observed_at":"2026-08-02T03:20:43.405143Z","submitted_at":"2024-11-25T06:17:23Z","title":"VidHal: Benchmarking Temporal Hallucinations in Vision LLMs","version":3},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-23T16:57:12.821916Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2411.16771"},"observation_digest":"sha256:c01cf2ba0bf6ddbd22d6f6598d39f71b851f2f6ebd4d8509249df1c3488be323","observation_id":"75f5f241-f1d3-4427-b93b-433224b632db","resolution":{"observed_at":"2026-05-23T16:58:12.002715Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-08-11T16:58:29.460476Z","title":"Video-bench: A comprehensive benchmark and toolkit for evaluating video-based large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.09582","last_updated":"2025-01-18T00:52:42Z","snapshot_observed_at":"2026-08-13T04:38:50.516969Z","submitted_at":"2024-12-12T18:54:48Z","title":"Neptune: The Long Orbit to Benchmarking Long Video Understanding","version":2},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-11T16:58:29.460476Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2412.09582"},"observation_digest":"sha256:08489899724211c6c60ab63772d5c8aae4a621167dab5beb8ac15cea483e3d20","observation_id":"e2b48806-8901-43d2-9c9c-310312ee36ff","resolution":{"observed_at":"2026-08-11T16:58:29.460476Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-08-11T16:56:06.311193Z","title":"Video-bench: A comprehensive benchmark and toolkit for evaluat- ing video-based large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.09596","last_updated":"2024-12-12T18:58:30Z","snapshot_observed_at":"2026-08-12T23:49:37.475723Z","submitted_at":"2024-12-12T18:58:30Z","title":"InternLM-XComposer2.5-OmniLive: A Comprehensive Multimodal System for Long-term Streaming Video and Audio Interactions","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-11T16:56:06.311193Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2412.09596"},"observation_digest":"sha256:112e43f59d160d7bdba3bb9294565894dd505e8adf96d642b3c9b22eff1bfea9","observation_id":"facc0219-2a24-4ea2-a415-bd5a42daf1f1","resolution":{"observed_at":"2026-08-11T16:56:06.311193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-08-11T14:22:51.012539Z","title":"Video-bench: A comprehensive benchmark and toolkit for evaluating video-based large language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.12075","last_updated":"2024-12-16T18:46:45Z","snapshot_observed_at":"2026-08-13T05:16:24.913588Z","submitted_at":"2024-12-16T18:46:45Z","title":"CG-Bench: Clue-grounded Question Answering Benchmark for Long Video Understanding","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-11T14:22:51.012539Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2412.12075"},"observation_digest":"sha256:aec146b0e08b44a36979764008f181e7378be54f3cce428cb34e8f58282ba6b4","observation_id":"0d49752f-6432-41a9-b1a4-dcbdc4b942ad","resolution":{"observed_at":"2026-08-11T14:22:51.012539Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":"2311.16103","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-07-04T06:39:37.654958Z","title":"Video-bench: A com- prehensive benchmark and toolkit for evaluating video-based 10 large language models","venue":null,"work_id":"37c46a44-ccc9-4891-a106-29a0fb8d3d27","year":2023},"citing_paper":{"arxiv_id":"2412.14171","last_updated":"2025-07-02T21:00:36Z","snapshot_observed_at":"2026-08-10T08:02:13.965614Z","submitted_at":"2024-12-18T18:59:54Z","title":"Thinking in Space: How Multimodal Large Language Models See, Remember, and Recall Spaces","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-22T09:27:43.919941Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2412.14171"},"observation_digest":"sha256:5f8ae898d93b8ed7be785ca968ef8b1fbd3fb06c9a7e80ae2dea111c4fce9abb","observation_id":"59d1c639-d981-4237-b57d-e4f8c904d5b8","resolution":{"observed_at":"2026-05-22T09:27:44.178010Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-08-11T11:09:13.328262Z","title":"Video-bench: A comprehensive benchmark and toolkit for evaluat- ing video-based large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.15838","last_updated":"2024-12-30T07:27:58Z","snapshot_observed_at":"2026-08-13T05:16:24.404304Z","submitted_at":"2024-12-20T12:27:16Z","title":"Align Anything: Training All-Modality Models to Follow Instructions with Language Feedback","version":2},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-11T11:09:13.328262Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2412.15838"},"observation_digest":"sha256:25052c44510af897b72d132382f2e20cea4455c20cfe1aa7b065e02db246e75b","observation_id":"65f5c485-f92e-4fb4-a78c-7f68d539b65a","resolution":{"observed_at":"2026-08-11T11:09:13.328262Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-08-11T05:24:17.780895Z","title":"Video-bench: A comprehensive benchmark and toolkit for evaluating video-based large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.17637","last_updated":"2024-12-23T15:13:56Z","snapshot_observed_at":"2026-08-13T05:16:25.777492Z","submitted_at":"2024-12-23T15:13:56Z","title":"SCBench: A Sports Commentary Benchmark for Video LLMs","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-11T05:24:17.780895Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2412.17637"},"observation_digest":"sha256:978052ba39d0701a887206d08536c50d0a4c0396360e0a9b5de67f30dc9e4bf9","observation_id":"790bf743-95bc-4de3-be39-e8ff36a81b80","resolution":{"observed_at":"2026-08-11T05:24:17.780895Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-08-10T22:19:53.440780Z","title":"Video-bench: A comprehen- sive benchmark and toolkit for evaluating video-based large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.02135","last_updated":"2025-01-03T23:03:24Z","snapshot_observed_at":"2026-08-13T22:51:03.927747Z","submitted_at":"2025-01-03T23:03:24Z","title":"AVTrustBench: Assessing and Enhancing Reliability and Robustness in Audio-Visual LLMs","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-10T22:19:53.440780Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2501.02135"},"observation_digest":"sha256:fde0d2febc92a276092e13c3d0fed4cee9adf6ab841c118a4fc7b0efb1ad3358","observation_id":"bbd32194-8fee-41c4-9768-d7b7d9e0d75e","resolution":{"observed_at":"2026-08-10T22:19:53.440780Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-08-10T17:15:23.823071Z","title":"Pan Lu, Hritik Bansal, Tony Xia, Jiacheng Liu, Chunyuan Li, Hannaneh Hajishirzi, Hao Cheng, Kai- Wei Chang, Michel Galley, and Jianfeng Gao","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2501.12380","last_updated":"2025-01-21T18:56:18Z","snapshot_observed_at":"2026-08-14T07:40:02.415562Z","submitted_at":"2025-01-21T18:56:18Z","title":"MMVU: Measuring Expert-Level Multi-Discipline Video Understanding","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-10T17:15:23.823071Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2501.12380"},"observation_digest":"sha256:b3eb793193c667e77d86f29afdd52e7400d9fa9a0447cad2a3c5eca2d27bcca6","observation_id":"d26d89f2-222c-48f3-b906-17ba2aab3661","resolution":{"observed_at":"2026-08-10T17:15:23.823071Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":"2311.16103","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-07-04T06:39:37.654958Z","title":"Video-bench: A com- prehensive benchmark and toolkit for evaluating video-based 10 large language models","venue":null,"work_id":"37c46a44-ccc9-4891-a106-29a0fb8d3d27","year":2023},"citing_paper":{"arxiv_id":"2501.13826","last_updated":"2025-01-23T16:51:47Z","snapshot_observed_at":"2026-07-06T20:25:03.950783Z","submitted_at":"2025-01-23T16:51:47Z","title":"Video-MMMU: Evaluating Knowledge Acquisition from Multi-Discipline Professional Videos","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-14T00:32:41.059558Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2501.13826"},"observation_digest":"sha256:56e0adb6756d688306e1127201244ece7dab825c73d93cd91c43d086900d75fd","observation_id":"c06a4d2c-520e-4fbb-aff2-933b4f80fb63","resolution":{"observed_at":"2026-05-14T00:32:41.171272Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":"2311.16103","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-07-04T06:39:37.654958Z","title":"Video-bench: A com- prehensive benchmark and toolkit for evaluating video-based 10 large language models","venue":null,"work_id":"37c46a44-ccc9-4891-a106-29a0fb8d3d27","year":2023},"citing_paper":{"arxiv_id":"2502.04326","last_updated":"2026-03-01T04:35:41Z","snapshot_observed_at":"2026-08-12T16:10:35.170069Z","submitted_at":"2025-02-06T18:59:40Z","title":"WorldSense: Evaluating Real-world Omnimodal Understanding for Multimodal LLMs","version":3},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-17T05:53:26.066674Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2502.04326"},"observation_digest":"sha256:1a11b0c6cfd39f85746766e1cfc37e5b7f24b2a22384955f67b5573606933c92","observation_id":"832a929d-004e-4f4f-9261-e952cac7cfb0","resolution":{"observed_at":"2026-05-17T05:53:26.388526Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-08-07T15:21:05.497735Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15435","last_updated":"2025-05-21T12:18:02Z","snapshot_observed_at":"2026-08-14T06:19:09.973445Z","submitted_at":"2025-05-21T12:18:02Z","title":"TimeCausality: Evaluating the Causal Ability in Time Dimension for Vision Language Models","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-07T15:21:05.497735Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2505.15435"},"observation_digest":"sha256:1e9a57eaffbc96c9b02ee5251f1ff7884299fb2c6b58ef50ecc9c63e33088dc8","observation_id":"c9c0aa41-0ace-4d52-82ef-682ec8b0919c","resolution":{"observed_at":"2026-08-07T15:21:05.497735Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-08-07T12:45:47.827826Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.23693","last_updated":"2025-05-29T17:31:13Z","snapshot_observed_at":"2026-08-12T14:38:08.894823Z","submitted_at":"2025-05-29T17:31:13Z","title":"VF-Eval: Evaluating Multimodal LLMs for Generating Feedback on AIGC Videos","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-07T12:45:47.827826Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2505.23693"},"observation_digest":"sha256:20d83b219089962d44ade919c91aedf7e8a22219d8794ef66c6de9b1dc220547","observation_id":"e79e1b7e-d904-4d9e-bac3-d687bd900045","resolution":{"observed_at":"2026-08-07T12:45:47.827826Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-08-07T12:35:18.562580Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.24257","last_updated":"2025-05-30T06:32:26Z","snapshot_observed_at":"2026-08-09T16:57:27.213555Z","submitted_at":"2025-05-30T06:32:26Z","title":"Out of Sight, Not Out of Context? Egocentric Spatial Reasoning in VLMs Across Disjoint Frames","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-07T12:35:18.562580Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2505.24257"},"observation_digest":"sha256:a88e71c86b90fc15c448578ab5e662d96aec9bbe0c335d9c0c7ba89f26cc95a7","observation_id":"64cc6b8b-9e79-4452-8c8f-43cbbc82b203","resolution":{"observed_at":"2026-08-07T12:35:18.562580Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-08-07T12:32:12.770007Z","title":"Video-bench: A comprehensive benchmark and toolkit for evaluating video-based large language models.arXiv preprint arXiv:2311.16103, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.24346","last_updated":"2025-05-30T08:39:36Z","snapshot_observed_at":"2026-08-14T06:46:41.679387Z","submitted_at":"2025-05-30T08:39:36Z","title":"VUDG: A Dataset for Video Understanding Domain Generalization","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T12:32:12.770007Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2505.24346"},"observation_digest":"sha256:18b99e1509e005217f176acd2a48687133f17b158dfda5dd4e7e3f8b27d3d71f","observation_id":"dec74c2f-daa6-4af3-993d-154d4e839773","resolution":{"observed_at":"2026-08-07T12:32:12.770007Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-08-07T10:50:52.887571Z","title":"Video-bench: A comprehensive benchmark and toolkit for evaluating video-based large language models.arXiv preprint arXiv:2311.16103, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05414","last_updated":"2025-06-04T19:11:20Z","snapshot_observed_at":"2026-08-12T02:45:52.278549Z","submitted_at":"2025-06-04T19:11:20Z","title":"SAVVY: Spatial Awareness via Audio-Visual LLMs through Seeing and Hearing","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T10:50:52.887571Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2506.05414"},"observation_digest":"sha256:606452e071e9747ace4e6526ee345218d4dcf30d323ab11ab7cfc3b0291fe180","observation_id":"7015e766-957a-4caa-ae1a-f5ee80c334fc","resolution":{"observed_at":"2026-08-07T10:50:52.887571Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":"2311.16103","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-07-04T06:39:37.654958Z","title":"Video-bench: A com- prehensive benchmark and toolkit for evaluating video-based 10 large language models","venue":null,"work_id":"37c46a44-ccc9-4891-a106-29a0fb8d3d27","year":2023},"citing_paper":{"arxiv_id":"2506.05425","last_updated":"2026-04-28T02:01:09Z","snapshot_observed_at":"2026-08-11T19:02:35.603239Z","submitted_at":"2025-06-05T05:51:35Z","title":"SIV-Bench: A Video Benchmark for Social Interaction Understanding and Reasoning","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-19T11:36:36.687324Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2506.05425"},"observation_digest":"sha256:97fb04e04d6029fbf2f2342cbd26ce94aada9fcf2a56c3afecdafc1184a84d90","observation_id":"6521a2b5-eede-41bd-af83-a54a6c289419","resolution":{"observed_at":"2026-05-19T11:37:15.771494Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-08-07T05:49:53.591063Z","title":"Video- bench: A comprehensive benchmark and toolkit for evaluating video-based large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.07016","last_updated":"2025-06-13T19:05:47Z","snapshot_observed_at":"2026-08-08T12:14:39.714543Z","submitted_at":"2025-06-08T06:34:29Z","title":"MAGNET: A Multi-agent Framework for Finding Audio-Visual Needles by Reasoning over Multi-Video Haystacks","version":2},"reference_index":120,"source":"pdf_text","source_observed_at":"2026-08-07T05:49:53.591063Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2506.07016"},"observation_digest":"sha256:cf060ac250c428ad031154dd7e7f050614de3b1359dadb95a788f96f1d559412","observation_id":"8a1f05c0-d91a-4c7a-830c-077a1a9b263c","resolution":{"observed_at":"2026-08-07T05:49:53.591063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-08-07T04:22:55.944763Z","title":"Video-bench: A com- prehensive benchmark and toolkit for evaluating video-based large language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.10857","last_updated":"2025-08-04T09:11:48Z","snapshot_observed_at":"2026-08-12T05:13:28.322748Z","submitted_at":"2025-06-12T16:17:17Z","title":"VRBench: A Benchmark for Multi-Step Reasoning in Long Narrative Videos","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T04:22:55.944763Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2506.10857"},"observation_digest":"sha256:877b3eb32497079e766ef98fd6d8a78ecc7ab092f2a6551e3aa159e630d3a40e","observation_id":"ad1ab0a5-8d56-42b7-abc7-9ab6202c0e16","resolution":{"observed_at":"2026-08-07T04:22:55.944763Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-08-06T21:59:16.696016Z","title":"arXiv preprint arXiv:2311.16103","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.02946","last_updated":"2025-06-28T15:24:05Z","snapshot_observed_at":"2026-08-11T08:23:42.830961Z","submitted_at":"2025-06-28T15:24:05Z","title":"Iterative Zoom-In: Temporal Interval Exploration for Long Video Understanding","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T21:59:16.696016Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2507.02946"},"observation_digest":"sha256:95e1326e634be527c18a2babacfb172f48a9d0812f3b8452360879f3964152d7","observation_id":"7d870a68-30b1-48c1-adc5-d6dd3459f69a","resolution":{"observed_at":"2026-08-06T21:59:16.696016Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-08-06T19:55:03.419558Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.04451","last_updated":"2025-07-06T16:17:32Z","snapshot_observed_at":"2026-08-09T23:14:43.081218Z","submitted_at":"2025-07-06T16:17:32Z","title":"CoT-lized Diffusion: Let's Reinforce T2I Generation Step-by-step","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T19:55:03.419558Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2507.04451"},"observation_digest":"sha256:b1d86de3d1ed68a1f55ac5f20ae418f6973c2cfafa6afeb378ffc5b28dca01cc","observation_id":"5ed3b0e7-ba83-42af-a584-0aec14638407","resolution":{"observed_at":"2026-08-06T19:55:03.419558Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-08-07T10:45:27.002204Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.08104","last_updated":"2025-06-04T21:58:50Z","snapshot_observed_at":"2026-08-13T15:23:36.449544Z","submitted_at":"2025-06-04T21:58:50Z","title":"VideoConviction: A Multimodal Benchmark for Human Conviction and Stock Market Recommendations","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T10:45:27.002204Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2507.08104"},"observation_digest":"sha256:a96083caa173f603784d5777cb47e0a31e203af722e2283c47b97685681bd2db","observation_id":"6fe0b637-3e4b-400a-9be1-0b50d4404bfb","resolution":{"observed_at":"2026-08-07T10:45:27.002204Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-08-06T17:58:27.742728Z","title":"arXiv preprint arXiv:2311.16103","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.09491","last_updated":"2025-07-13T04:44:57Z","snapshot_observed_at":"2026-08-13T19:33:23.185466Z","submitted_at":"2025-07-13T04:44:57Z","title":"GLIMPSE: Do Large Vision-Language Models Truly Think With Videos or Just Glimpse at Them?","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T17:58:27.742728Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2507.09491"},"observation_digest":"sha256:9cf458c51e6e38c70b8c96d0b4173adaa448757adc4d345ea30eb25cee82e05c","observation_id":"e3afeb3f-c265-4265-b6d6-c827bc88199f","resolution":{"observed_at":"2026-08-06T17:58:27.742728Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-08-06T05:15:42.629926Z","title":"Video-bench: A comprehensive benchmark and toolkit for evaluat- ing video-based large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.02094","last_updated":"2025-08-04T06:05:36Z","snapshot_observed_at":"2026-08-08T15:06:07.047501Z","submitted_at":"2025-08-04T06:05:36Z","title":"\"Harmless to You, Hurtful to Me!\": Investigating the Detection of Toxic Languages Grounded in the Perspective of Youth","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-06T05:15:42.629926Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2508.02094"},"observation_digest":"sha256:b6c82eff3cf9eec37f380f87b7dec05f2340b38cefc3471d321e83e364083579","observation_id":"ee9acb31-5a21-4045-b2e0-84dd3f2f04f2","resolution":{"observed_at":"2026-08-06T05:15:42.629926Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-08-06T05:12:37.260203Z","title":"Video-bench: A comprehensive benchmark and toolkit for evaluating video-based large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.02095","last_updated":"2025-08-06T19:21:50Z","snapshot_observed_at":"2026-08-08T15:06:09.808299Z","submitted_at":"2025-08-04T06:06:06Z","title":"VLM4D: Towards Spatiotemporal Awareness in Vision Language Models","version":2},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-08-06T05:12:37.260203Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2508.02095"},"observation_digest":"sha256:e3bd81413ce3ff891ddaa2ed092cf256fb9486e7f72fb769a9f1dceabefc1fa2","observation_id":"67684500-24fc-4457-a575-c13145b72f96","resolution":{"observed_at":"2026-08-06T05:12:37.260203Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-08-06T04:49:40.297857Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.03039","last_updated":"2025-08-05T03:33:24Z","snapshot_observed_at":"2026-08-13T14:57:48.174747Z","submitted_at":"2025-08-05T03:33:24Z","title":"VideoForest: Person-Anchored Hierarchical Reasoning for Cross-Video Question Answering","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T04:49:40.297857Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2508.03039"},"observation_digest":"sha256:e39a166580a0f67111055652255510a10bd6e5f38085490688ed6a89e0b6bab2","observation_id":"bfe8e4e3-9465-4421-9f4b-1e4d1c9e580b","resolution":{"observed_at":"2026-08-06T04:49:40.297857Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-08-04T20:32:56.736581Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.08538","last_updated":"2025-09-11T11:14:00Z","snapshot_observed_at":"2026-08-13T02:52:28.973488Z","submitted_at":"2025-09-10T12:34:07Z","title":"MESH -- Understanding Videos Like Human: Measuring Hallucinations in Large Video Models","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-04T20:32:56.736581Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2509.08538"},"observation_digest":"sha256:aeedb1d224024ad0132a49722bca696186a345558bc4268021585fa6b57782ae","observation_id":"21a2c4cd-3046-4363-94bf-87e42334fa85","resolution":{"observed_at":"2026-08-04T20:32:56.736581Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-08-04T20:20:36.840834Z","title":"Video-bench: A com- prehensive benchmark and toolkit for evaluating video-based large language models.CoRR, abs/2311.16103, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.08621","last_updated":"2025-09-10T14:17:53Z","snapshot_observed_at":"2026-08-09T06:09:37.758584Z","submitted_at":"2025-09-10T14:17:53Z","title":"AdsQA: Towards Advertisement Video Understanding","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-04T20:20:36.840834Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2509.08621"},"observation_digest":"sha256:313aa1f6ea232aeaa87f5964c2065ba0789325d6124044feaea52679fd8b3bb3","observation_id":"f481b5b8-2317-4325-bedb-abeca7dfaecd","resolution":{"observed_at":"2026-08-04T20:20:36.840834Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-08-04T13:54:19.068356Z","title":", Zhu , B","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.24563","last_updated":"2026-07-15T12:54:15Z","snapshot_observed_at":"2026-08-10T11:27:30.454931Z","submitted_at":"2025-09-29T10:16:05Z","title":"NeMo: Needle in a Montage for Video-Language Understanding","version":3},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-04T13:54:19.068356Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2509.24563"},"observation_digest":"sha256:e36f182276df76e1f332d011fcb88fecf7b97b2f46e0db6171875d49177871f7","observation_id":"ab2acae2-096e-49db-a432-7471022ac7b0","resolution":{"observed_at":"2026-08-04T13:54:19.068356Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":"2311.16103","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-07-04T06:39:37.654958Z","title":"Video-bench: A com- prehensive benchmark and toolkit for evaluating video-based 10 large language models","venue":null,"work_id":"37c46a44-ccc9-4891-a106-29a0fb8d3d27","year":2023},"citing_paper":{"arxiv_id":"2511.00503","last_updated":"2026-04-07T13:15:23Z","snapshot_observed_at":"2026-08-11T11:31:29.982561Z","submitted_at":"2025-11-01T11:16:25Z","title":"Diff4Splat: Controllable 4D Scene Generation with Latent Dynamic Reconstruction Models","version":2},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-05-18T01:59:23.928725Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2511.00503"},"observation_digest":"sha256:ed8c5b4e6634004fbedaabdc8634fbc063d2012adb04b9b724540d373d9c76a1","observation_id":"f8fb1a83-acef-479f-a08f-2d2b8d26c3f7","resolution":{"observed_at":"2026-05-18T02:00:39.537569Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":"2311.16103","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-07-04T06:39:37.654958Z","title":"Video-bench: A com- prehensive benchmark and toolkit for evaluating video-based 10 large language models","venue":null,"work_id":"37c46a44-ccc9-4891-a106-29a0fb8d3d27","year":2023},"citing_paper":{"arxiv_id":"2512.13281","last_updated":"2026-05-07T09:33:52Z","snapshot_observed_at":"2026-08-13T09:58:32.320290Z","submitted_at":"2025-12-15T12:41:23Z","title":"VideoASMR-Bench: Can AI-Generated ASMR Videos Fool VLMs and Humans?","version":4},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-16T21:46:43.353305Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2512.13281"},"observation_digest":"sha256:ca29b7a5956db5b75b9f87922a7063d3309a18d00ee11e05ee0fd272015e37b9","observation_id":"fd65e0bd-1307-4f2b-8dca-5361e2f58090","resolution":{"observed_at":"2026-05-16T21:48:34.548699Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":"2311.16103","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-07-04T06:39:37.654958Z","title":"Video-bench: A com- prehensive benchmark and toolkit for evaluating video-based 10 large language models","venue":null,"work_id":"37c46a44-ccc9-4891-a106-29a0fb8d3d27","year":2023},"citing_paper":{"arxiv_id":"2601.15724","last_updated":"2026-04-19T09:17:00Z","snapshot_observed_at":"2026-08-13T23:11:17.607349Z","submitted_at":"2026-01-22T07:47:29Z","title":"VideoThinker: Building Agentic VideoLLMs with LLM-Guided Tool Reasoning","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-16T12:17:42.135851Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2601.15724"},"observation_digest":"sha256:877142080353e8362c9dc765809388374f567226ae7aae3e742217ec0dd4c8d8","observation_id":"472bbbbf-c89c-4212-a656-713ea518a437","resolution":{"observed_at":"2026-05-16T12:17:51.900773Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":"2311.16103","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-07-04T06:39:37.654958Z","title":"Video-bench: A com- prehensive benchmark and toolkit for evaluating video-based 10 large language models","venue":null,"work_id":"37c46a44-ccc9-4891-a106-29a0fb8d3d27","year":2023},"citing_paper":{"arxiv_id":"2604.06339","last_updated":"2026-04-07T18:17:05Z","snapshot_observed_at":"2026-07-06T22:54:53.308309Z","submitted_at":"2026-04-07T18:17:05Z","title":"Evolution of Video Generative Foundations","version":1},"reference_index":183,"source":"pdf_text","source_observed_at":"2026-05-10T18:41:38.616611Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2604.06339"},"observation_digest":"sha256:ae07a4db41f91ded64433fd57d03dc2d916437e09da0dd94898fbd0b799f1376","observation_id":"b65ff505-7262-4668-8e46-ac9ac36a94b7","resolution":{"observed_at":"2026-05-11T00:05:51.688660Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":"2311.16103","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-07-04T06:39:37.654958Z","title":"Video-bench: A com- prehensive benchmark and toolkit for evaluating video-based 10 large language models","venue":null,"work_id":"37c46a44-ccc9-4891-a106-29a0fb8d3d27","year":2023},"citing_paper":{"arxiv_id":"2604.22884","last_updated":"2026-04-24T08:13:19Z","snapshot_observed_at":"2026-08-11T13:22:00.434325Z","submitted_at":"2026-04-24T08:13:19Z","title":"Can Multimodal Large Language Models Truly Understand Small Objects?","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-08T12:49:53.645987Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2604.22884"},"observation_digest":"sha256:5a9cc76220e34ad3bae1e0e568c1a064a5e25949ca723b8b82b0228bc5957b55","observation_id":"a5d7c21a-0f95-471f-8249-cbd7ba60f8a8","resolution":{"observed_at":"2026-05-11T19:01:19.368825Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":"2311.16103","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-07-04T06:39:37.654958Z","title":"Video-bench: A com- prehensive benchmark and toolkit for evaluating video-based 10 large language models","venue":null,"work_id":"37c46a44-ccc9-4891-a106-29a0fb8d3d27","year":2023},"citing_paper":{"arxiv_id":"2605.17283","last_updated":"2026-05-17T06:39:05Z","snapshot_observed_at":"2026-08-08T15:14:20.631028Z","submitted_at":"2026-05-17T06:39:05Z","title":"OProver: A Unified Framework for Agentic Formal Theorem Proving","version":1},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-05-20T14:43:46.517807Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2605.17283"},"observation_digest":"sha256:27ff2b48ae9307786412a3cecddd3ec1b4e5f77d253eb57ed6b823f0cfd802e9","observation_id":"e7457510-7bcc-4a34-abf6-c7bef05eca4e","resolution":{"observed_at":"2026-05-20T14:48:23.414269Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","version":2},"cited_work":{"arxiv_id":"2311.16103","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.16103","snapshot_observed_at":"2026-07-04T06:39:37.654958Z","title":"Video-bench: A com- prehensive benchmark and toolkit for evaluating video-based 10 large language models","venue":null,"work_id":"37c46a44-ccc9-4891-a106-29a0fb8d3d27","year":2023},"citing_paper":{"arxiv_id":"2606.21734","last_updated":"2026-06-19T20:43:49Z","snapshot_observed_at":"2026-08-05T18:05:51.515234Z","submitted_at":"2026-06-19T20:43:49Z","title":"HPP: Hierarchical Programmatic Probing for Long Video Understanding by Decoupling Perception and Reasoning","version":1},"reference_index":183,"source":"arxiv_source","source_observed_at":"2026-06-26T14:19:53.450263Z"},"links":{"cited_paper":"/paper/2311.16103","citing_paper":"/paper/2606.21734"},"observation_digest":"sha256:8698362a66747989295638bf62aeba602af5c20a91b5b5493f24ec7cbcd546f0","observation_id":"54501be4-b06e-4906-b391-dc0aef2ece7b","resolution":{"observed_at":"2026-07-04T06:39:37.656222Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2311.16103/citation-record","integrity":"/paper/2311.16103/integrity","json":"/paper/2311.16103/citation-record.json","paper":"/paper/2311.16103"},"outbound":[],"paper":{"arxiv_id":"2311.16103","last_updated":"2023-11-28T18:16:29Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-13T05:16:05.220753Z","submitted_at":"2023-11-27T18:59:58Z","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"thesis":"As of 14 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 42 inbound Pith citation observations for arXiv:2311.16103."}