{"as_of":"2026-08-10T04:12:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:eb5eda242e7f95b62b902a487a9698ce33c178fa4550dfa1fe93fb17aa5ed2d5","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":18,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":18,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":18,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":18,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:34:43.788047Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-02T12:26:57.184700Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":"2406.11303","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-07-02T12:26:57.184700Z","title":"Videovista: A versatile bench- mark for video understanding and reasoning","venue":null,"work_id":"b5b7e620-228d-4c25-9017-df5c7b62a760","year":2024},"citing_paper":{"arxiv_id":"2501.02955","last_updated":"2026-05-12T15:02:48Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-06T11:57:38Z","title":"MotionBench: Benchmarking and Improving Fine-grained Video Motion Understanding for Vision Language Models","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-23T05:44:31.546843Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2501.02955"},"observation_digest":"sha256:465496bf2d8174a2b660ba0c8130cf104964b02b4fdd7ba487ec11edddc04977","observation_id":"df47b175-ed30-402b-9ad1-4104542abff3","resolution":{"observed_at":"2026-05-23T05:45:28.342097Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-08-07T15:34:43.788047Z","title":"Videovista: A versatile benchmark for video understanding and reasoning.arXiv preprint arXiv:2406.11303, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14640","last_updated":"2025-05-20T17:26:32Z","snapshot_observed_at":"2026-08-07T15:28:25.992115Z","submitted_at":"2025-05-20T17:26:32Z","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T15:34:43.788047Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2505.14640"},"observation_digest":"sha256:3cee0a0d7ab97f77aa63d6eb98ae8e03e159fe491573636e28df3d7bd3d0a5ae","observation_id":"29aaa453-09ca-4625-84b3-d9457014c874","resolution":{"observed_at":"2026-08-07T15:34:43.788047Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-08-07T14:03:01.131871Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20124","last_updated":"2025-05-27T12:10:27Z","snapshot_observed_at":"2026-08-08T09:10:32.146078Z","submitted_at":"2025-05-26T15:24:06Z","title":"TUNA: Comprehensive Fine-grained Temporal Understanding Evaluation on Dense Dynamic Videos","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-07T14:03:01.131871Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2505.20124"},"observation_digest":"sha256:320e75f97a4e31bb9be3a0ab2798d9ec61df8c29630d2237ae44e38ea688343c","observation_id":"fc66876d-c16b-4d4b-977f-7b397612597f","resolution":{"observed_at":"2026-08-07T14:03:01.131871Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-08-07T12:48:03.738031Z","title":"Videovista: A versatile benchmark for video understanding and reasoning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23484","last_updated":"2025-05-29T14:34:25Z","snapshot_observed_at":"2026-08-09T05:32:01.732887Z","submitted_at":"2025-05-29T14:34:25Z","title":"VCapsBench: A Large-scale Fine-grained Benchmark for Video Caption Quality Evaluation","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T12:48:03.738031Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2505.23484"},"observation_digest":"sha256:5262d96e9c221a7bda71c1bc2fada08028c0b0d426508857d5bb8c731024ac6a","observation_id":"7514bf63-6fd6-4d1f-91bb-5fac0532564c","resolution":{"observed_at":"2026-08-07T12:48:03.738031Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-08-07T12:45:46.866771Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23693","last_updated":"2025-05-29T17:31:13Z","snapshot_observed_at":"2026-08-07T18:41:28.769570Z","submitted_at":"2025-05-29T17:31:13Z","title":"VF-Eval: Evaluating Multimodal LLMs for Generating Feedback on AIGC Videos","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-07T12:45:46.866771Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2505.23693"},"observation_digest":"sha256:467cbe87d02b6cc6b989dbeed12a1553183c94b7e4e5d4ca02bed7ea9f8a58c9","observation_id":"f4e62474-1e58-4261-b56c-11dbe3e6481f","resolution":{"observed_at":"2026-08-07T12:45:46.866771Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-08-07T12:40:44.943565Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23922","last_updated":"2025-05-29T18:15:07Z","snapshot_observed_at":"2026-08-09T06:58:13.030008Z","submitted_at":"2025-05-29T18:15:07Z","title":"ScaleLong: A Multi-Timescale Benchmark for Long Video Understanding","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-07T12:40:44.943565Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2505.23922"},"observation_digest":"sha256:9eff59753e9c296ca3eadd9447daeda574f3a54fa86dfaf2b44eb86f1eafb9b7","observation_id":"6f5a593c-f7dc-4764-85f5-da4346e18514","resolution":{"observed_at":"2026-08-07T12:40:44.943565Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-08-07T12:32:10.535033Z","title":"Videovista: A versatile benchmark for video understanding and reasoning.arXiv preprint arXiv:2406.11303, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24346","last_updated":"2025-05-30T08:39:36Z","snapshot_observed_at":"2026-08-09T11:59:19.522063Z","submitted_at":"2025-05-30T08:39:36Z","title":"VUDG: A Dataset for Video Understanding Domain Generalization","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T12:32:10.535033Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2505.24346"},"observation_digest":"sha256:28a5b3b174e37db7e2a0d7c71ce3b0769c0e11ffa17211d75dff5ee859216fbb","observation_id":"408e7f61-f69b-46cd-97ec-da0311486d57","resolution":{"observed_at":"2026-08-07T12:32:10.535033Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":"2406.11303","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-07-02T12:26:57.184700Z","title":"Videovista: A versatile bench- mark for video understanding and reasoning","venue":null,"work_id":"b5b7e620-228d-4c25-9017-df5c7b62a760","year":2024},"citing_paper":{"arxiv_id":"2506.05425","last_updated":"2026-04-28T02:01:09Z","snapshot_observed_at":"2026-07-31T07:37:26.215945Z","submitted_at":"2025-06-05T05:51:35Z","title":"SIV-Bench: A Video Benchmark for Social Interaction Understanding and Reasoning","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-19T11:36:36.687324Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2506.05425"},"observation_digest":"sha256:92eac91f466a8f75b781b4104a22ef99306ac7234e38304b2085edaa442e229c","observation_id":"f72b07e1-89ec-4e63-a23d-54c6a082e19c","resolution":{"observed_at":"2026-05-19T11:37:15.739001Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-08-07T04:22:55.901902Z","title":"Videovista: A versatile bench- mark for video understanding and reasoning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10857","last_updated":"2025-08-04T09:11:48Z","snapshot_observed_at":"2026-08-08T13:30:57.844783Z","submitted_at":"2025-06-12T16:17:17Z","title":"VRBench: A Benchmark for Multi-Step Reasoning in Long Narrative Videos","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T04:22:55.901902Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2506.10857"},"observation_digest":"sha256:962a0396eb87089f840e3e137f6ab285e980d541d131fea237f46438b07e9387","observation_id":"72944cf2-74c2-4dd2-91fa-073fcf1672fa","resolution":{"observed_at":"2026-08-07T04:22:55.901902Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-08-07T00:41:51.290582Z","title":"Videovista: A versatile bench- mark for video understanding and reasoning.arXiv preprint arXiv:2406.11303, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12992","last_updated":"2025-06-15T23:20:08Z","snapshot_observed_at":"2026-08-07T00:35:03.776740Z","submitted_at":"2025-06-15T23:20:08Z","title":"SmartHome-Bench: A Comprehensive Benchmark for Video Anomaly Detection in Smart Homes Using Multi-Modal Large Language Models","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T00:41:51.290582Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2506.12992"},"observation_digest":"sha256:fd1bce31165cffae6bfe634ff7547a3efa91a9c10045c173120eb78ca4ddf664","observation_id":"fd3247c8-5278-4a34-86de-a9e8723a234c","resolution":{"observed_at":"2026-08-07T00:41:51.290582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-08-06T04:49:39.159644Z","title":"https://api.semanticscholar.org/CorpusID: 270559556","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.03039","last_updated":"2025-08-05T03:33:24Z","snapshot_observed_at":"2026-08-07T14:34:47.687957Z","submitted_at":"2025-08-05T03:33:24Z","title":"VideoForest: Person-Anchored Hierarchical Reasoning for Cross-Video Question Answering","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-06T04:49:39.159644Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2508.03039"},"observation_digest":"sha256:3a5dcff4c680904b003b261d4b8a4455848cda3aecdf2b2ae5874a74e76a9538","observation_id":"7d166b95-e89a-4822-9a73-8309863d4e2a","resolution":{"observed_at":"2026-08-06T04:49:39.159644Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-08-04T13:54:15.574763Z","title":", Chen , X","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.24563","last_updated":"2026-07-15T12:54:15Z","snapshot_observed_at":"2026-08-08T23:25:33.156446Z","submitted_at":"2025-09-29T10:16:05Z","title":"NeMo: Needle in a Montage for Video-Language Understanding","version":3},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-04T13:54:15.574763Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2509.24563"},"observation_digest":"sha256:28a18e71814d8f213483fc00e051cb4ab56ba3ae0afe8b9fa666f7bd70f19d16","observation_id":"4880c9fe-dc2e-4103-adad-2d91bd29d674","resolution":{"observed_at":"2026-08-04T13:54:15.574763Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":"2406.11303","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-07-02T12:26:57.184700Z","title":"Videovista: A versatile bench- mark for video understanding and reasoning","venue":null,"work_id":"b5b7e620-228d-4c25-9017-df5c7b62a760","year":2024},"citing_paper":{"arxiv_id":"2603.27259","last_updated":"2026-06-18T21:01:40Z","snapshot_observed_at":"2026-08-02T11:52:58.572026Z","submitted_at":"2026-03-28T12:44:19Z","title":"Seeing the Scene Matters: Revealing Forgetting in Video Understanding Models with a Scene-Aware Long-Video Benchmark","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-14T22:05:07.326202Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2603.27259"},"observation_digest":"sha256:86c64edbc1fe6f2103292765d8ab45b84298dbbb4fde0f951f955a9d6232491d","observation_id":"4f18c017-fa11-49b1-9ff5-5a40e6d0cc64","resolution":{"observed_at":"2026-05-14T22:08:04.406514Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":"2406.11303","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-07-02T12:26:57.184700Z","title":"Videovista: A versatile bench- mark for video understanding and reasoning","venue":null,"work_id":"b5b7e620-228d-4c25-9017-df5c7b62a760","year":2024},"citing_paper":{"arxiv_id":"2605.15342","last_updated":"2026-05-14T19:12:20Z","snapshot_observed_at":"2026-07-06T23:26:37.362117Z","submitted_at":"2026-05-14T19:12:20Z","title":"Minerva-Ego: Spatiotemporal Hints for Egocentric Video Understanding","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-19T16:02:53.887605Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2605.15342"},"observation_digest":"sha256:da10033498f91cf55f2e018856d876028392328fe5541518361550a53ea837d8","observation_id":"c02b375f-471e-495f-ba2d-dd8d1b12bdb0","resolution":{"observed_at":"2026-05-19T16:03:08.051487Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":"2406.11303","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-07-02T12:26:57.184700Z","title":"Videovista: A versatile bench- mark for video understanding and reasoning","venue":null,"work_id":"b5b7e620-228d-4c25-9017-df5c7b62a760","year":2024},"citing_paper":{"arxiv_id":"2606.03635","last_updated":"2026-06-02T13:31:57Z","snapshot_observed_at":"2026-08-05T07:22:25.874494Z","submitted_at":"2026-06-02T13:31:57Z","title":"VidMsg: A Benchmark for Implicit Message Inference in Short Videos","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-28T10:25:06.594946Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2606.03635"},"observation_digest":"sha256:d955d281e51c0a1dd7695cc9df5ffb06ef4e1e8c512ee26d27963f00144cec97","observation_id":"b7b3ffe2-8116-4a87-b31b-d0d2b8287e22","resolution":{"observed_at":"2026-07-02T02:56:30.085934Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":"2406.11303","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-07-02T12:26:57.184700Z","title":"Videovista: A versatile bench- mark for video understanding and reasoning","venue":null,"work_id":"b5b7e620-228d-4c25-9017-df5c7b62a760","year":2024},"citing_paper":{"arxiv_id":"2606.06338","last_updated":"2026-06-04T16:12:43Z","snapshot_observed_at":"2026-07-06T23:46:10.612537Z","submitted_at":"2026-06-04T16:12:43Z","title":"StoryVideoQA: Scaling Deep Video Understanding with a Large-Scale, Multi-Genre and Auto-Generated Dataset","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-06-28T02:05:47.810096Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2606.06338"},"observation_digest":"sha256:1cdaab3e99061d25335b73f6a384ffe82f26e776cd33390004118f8ea37ebb56","observation_id":"fa3a6b1c-d6de-40ef-bfcd-2f77df218b05","resolution":{"observed_at":"2026-07-02T12:26:57.186187Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":"2406.11303","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-07-02T12:26:57.184700Z","title":"Videovista: A versatile bench- mark for video understanding and reasoning","venue":null,"work_id":"b5b7e620-228d-4c25-9017-df5c7b62a760","year":2024},"citing_paper":{"arxiv_id":"2606.28593","last_updated":"2026-06-26T20:38:04Z","snapshot_observed_at":"2026-07-07T00:02:38.226622Z","submitted_at":"2026-06-26T20:38:04Z","title":"Animation2Code: Evaluating Temporal Visual Reasoning in Video-to-Code Generation","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-30T00:53:52.724902Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2606.28593"},"observation_digest":"sha256:6bde4a7ee6b4c78a2f17e46eca4a184f69ef2f05ff9c134854f96a34c22c1700","observation_id":"7ad99eda-7280-48fb-a4c6-8582ee1d7c21","resolution":{"observed_at":"2026-07-01T16:05:49.750946Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11303","snapshot_observed_at":"2026-08-07T04:24:54.981358Z","title":"arXiv preprint arXiv:2406.11303 , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.06361","last_updated":"2026-08-06T17:57:06Z","snapshot_observed_at":"2026-08-09T23:13:31.191909Z","submitted_at":"2026-08-06T17:57:06Z","title":"The Low Frequency Trap: Video Language Models Fail at Simple Event Bookkeeping","version":1},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-08-07T04:24:54.981358Z"},"links":{"cited_paper":"/paper/2406.11303","citing_paper":"/paper/2608.06361"},"observation_digest":"sha256:bbccb64d39b0dff1b820df1c042f626c84ef103894131887bd7482a9ff906f7b","observation_id":"e059e240-3680-4a48-8baa-dbd830e98b72","resolution":{"observed_at":"2026-08-07T04:24:54.981358Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2406.11303/citation-record","integrity":"/paper/2406.11303/integrity","json":"/paper/2406.11303/citation-record.json","paper":"/paper/2406.11303"},"outbound":[],"paper":{"arxiv_id":"2406.11303","last_updated":"2024-06-17T08:09:00Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-04T15:53:11.168523Z","submitted_at":"2024-06-17T08:09:00Z","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 18 inbound Pith citation observations for arXiv:2406.11303."}