{"as_of":"2026-08-09T19:50:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:f1a064277394a798d64762d87359fa04648e0745ed93396d178460f3f527c1c4","coverage":[{"denominator":59,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":59,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T17:49:34.963022Z","state":"measured"},{"denominator":63,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":63,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":4,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":4,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-28T01:52:44.785582Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-02T12:46:56.833017Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"cited_work":{"arxiv_id":"2507.09876","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.09876","snapshot_observed_at":"2026-07-02T12:46:56.833017Z","title":"arXiv preprint arXiv:2507.09876 , year=","venue":null,"work_id":"d08ca88f-8f20-4d09-880f-0151d0e8d1ba","year":2025},"citing_paper":{"arxiv_id":"2511.04570","last_updated":"2026-04-07T09:55:11Z","snapshot_observed_at":"2026-07-30T01:17:08.732833Z","submitted_at":"2025-11-06T17:25:23Z","title":"Thinking with Video: Video Generation as a Promising Multimodal Reasoning Paradigm","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-18T00:54:24.641649Z"},"links":{"cited_paper":"/paper/2507.09876","citing_paper":"/paper/2511.04570"},"observation_digest":"sha256:2f13e6fe89b262d2c8243b5db57f963dac90528e4298e7d452e7f275234f6958","observation_id":"558a108b-cffe-40fe-a71d-db26eebee2c9","resolution":{"observed_at":"2026-05-18T00:55:35.070077Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"cited_work":{"arxiv_id":"2507.09876","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.09876","snapshot_observed_at":"2026-07-02T12:46:56.833017Z","title":"arXiv preprint arXiv:2507.09876 , year=","venue":null,"work_id":"d08ca88f-8f20-4d09-880f-0151d0e8d1ba","year":2025},"citing_paper":{"arxiv_id":"2605.01657","last_updated":"2026-05-03T00:52:51Z","snapshot_observed_at":"2026-08-08T14:11:45.861701Z","submitted_at":"2026-05-03T00:52:51Z","title":"Act2See: Emergent Active Visual Perception for Video Reasoning","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-08T19:34:53.683729Z"},"links":{"cited_paper":"/paper/2507.09876","citing_paper":"/paper/2605.01657"},"observation_digest":"sha256:d3296b0ca3bc7846192e7308171df5d3633385b9d0452a8de4f79522cfda8635","observation_id":"51a2c363-4137-4637-9b3a-ed92ff0003fd","resolution":{"observed_at":"2026-05-09T05:45:22.948194Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"cited_work":{"arxiv_id":"2507.09876","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.09876","snapshot_observed_at":"2026-07-02T12:46:56.833017Z","title":"arXiv preprint arXiv:2507.09876 , year=","venue":null,"work_id":"d08ca88f-8f20-4d09-880f-0151d0e8d1ba","year":2025},"citing_paper":{"arxiv_id":"2605.17283","last_updated":"2026-05-17T06:39:05Z","snapshot_observed_at":"2026-08-08T15:14:20.631028Z","submitted_at":"2026-05-17T06:39:05Z","title":"OProver: A Unified Framework for Agentic Formal Theorem Proving","version":1},"reference_index":104,"source":"arxiv_source","source_observed_at":"2026-05-20T14:43:46.517807Z"},"links":{"cited_paper":"/paper/2507.09876","citing_paper":"/paper/2605.17283"},"observation_digest":"sha256:c9b82a06c697a9c73827ce9d408fdf6f949051565135ce26b1266d85e5ddead6","observation_id":"05811938-693a-4663-9912-8d5666776b84","resolution":{"observed_at":"2026-05-20T14:48:23.408448Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"cited_work":{"arxiv_id":"2507.09876","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.09876","snapshot_observed_at":"2026-07-02T12:46:56.833017Z","title":"arXiv preprint arXiv:2507.09876 , year=","venue":null,"work_id":"d08ca88f-8f20-4d09-880f-0151d0e8d1ba","year":2025},"citing_paper":{"arxiv_id":"2606.05736","last_updated":"2026-06-04T05:55:15Z","snapshot_observed_at":"2026-07-06T23:45:42.379051Z","submitted_at":"2026-06-04T05:55:15Z","title":"VTI-CoT: Visual-Textual Interleaved Chain of Thought for Video Reasoning","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-06-28T01:52:44.785582Z"},"links":{"cited_paper":"/paper/2507.09876","citing_paper":"/paper/2606.05736"},"observation_digest":"sha256:d68ff3272f4299aae8be29ec0af0dac9e64057f75b98b6ea7bfd5b0d0ee051c7","observation_id":"aed15bde-fead-4786-81d4-3742e7779b81","resolution":{"observed_at":"2026-07-02T12:46:56.834326Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2507.09876/citation-record","integrity":"/paper/2507.09876/integrity","json":"/paper/2507.09876/citation-record.json","paper":"/paper/2507.09876"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-06T17:49:34.724807Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.724807Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:9461b1267cdeb8a43a787131e0cf6311dd44a02732bcb7cadc24912b2dd66035","observation_id":"7cc29dba-3559-498c-aca7-61e5c70afd2c","resolution":{"observed_at":"2026-08-06T17:49:34.724807Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:39.360226Z","title":"Hadzic, Taran Kota, Jimming He, Cristobal Eyzaguirre, Zane Durante, Manling Li, Jiajun Wu, and Li Fei-Fei","venue":null,"work_id":"b3b73293-0410-4cd9-bde7-71c615144615","year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.729419Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:68ae4eabcbec6641c920659c5ef7672384331edac073831669af9514261177de","observation_id":"27dccd33-300a-42ae-bdef-f6fc5854b9cf","resolution":{"observed_at":"2026-08-06T17:49:39.444946Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:39.212057Z","title":null,"venue":null,"work_id":"30974041-ee16-4b7c-85fd-8165cbb26cf7","year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.734176Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:767d63a70dd2957789f5eb1e631f84e07627090597ba0c1fcb5db6181a6dcd7b","observation_id":"015d76b4-6404-43c0-8a9a-186d849caf1f","resolution":{"observed_at":"2026-08-06T17:49:39.276953Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09567","last_updated":"2025-07-18T15:57:54Z","snapshot_observed_at":"2026-08-08T22:33:20.124926Z","submitted_at":"2025-03-12T17:35:03Z","title":"Towards Reasoning Era: A Survey of Long Chain-of-Thought for Reasoning Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09567","snapshot_observed_at":"2026-08-06T17:49:34.738049Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.738049Z"},"links":{"cited_paper":"/paper/2503.09567","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:d645712d8c8d74b63331bde73ee3624dcd50e736af6e6e812941a141ae0fda38","observation_id":"9f229cfb-9c15-459d-ba5f-197b93d0830e","resolution":{"observed_at":"2026-08-06T17:49:34.738049Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:39.029215Z","title":null,"venue":null,"work_id":"55bf8c91-fc28-418a-af82-2a71afcac42e","year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.742594Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:b7bc42d302f0d3bd58a423fd7f931929096729ebb14a25eec1736c6e6f0c5e2f","observation_id":"27225dd8-3d9e-4ee7-ba66-f364c93899cb","resolution":{"observed_at":"2026-08-06T17:49:39.124749Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.01903","last_updated":"2025-08-05T16:19:40Z","snapshot_observed_at":"2026-08-08T04:30:09.801268Z","submitted_at":"2025-07-02T17:19:20Z","title":"AI4Research: A Survey of Artificial Intelligence for Scientific Research","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.01903","snapshot_observed_at":"2026-08-06T17:49:34.746783Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.746783Z"},"links":{"cited_paper":"/paper/2507.01903","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:6f236fafbdb3dca0a194e17ac568d54d05f64bd82c924af5326e7de6054bb6f4","observation_id":"6b51a282-7bef-46aa-bf45-3d01fdc5038b","resolution":{"observed_at":"2026-08-06T17:49:34.746783Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:34.751721Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.751721Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:bda846cae2d4ff0d6ab27bcbec319f6aef3e6d1c4c4d1d80f8da9e94626492ec","observation_id":"381c4fb1-79fd-43e4-874f-f9ace96b357c","resolution":{"observed_at":"2026-08-06T17:49:34.751721Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.11623","last_updated":"2024-10-15T14:08:53Z","snapshot_observed_at":"2026-07-06T19:33:52.737083Z","submitted_at":"2024-10-15T14:08:53Z","title":"VidEgoThink: Assessing Egocentric Video Understanding Capabilities for Embodied AI","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.11623","snapshot_observed_at":"2026-08-06T17:49:34.756138Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.756138Z"},"links":{"cited_paper":"/paper/2410.11623","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:49a0951c6a78a088438c2a92521de872a5064ccdd5540c6bf99a660ff49d2460","observation_id":"7cbc93df-fee0-4c19-be81-59e670680f03","resolution":{"observed_at":"2026-08-06T17:49:34.756138Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:34.760649Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.760649Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:384715307247d2415f2adf36aae923293157bb46ace6b9ae0a64056abbb097c6","observation_id":"2d461dc0-9426-464a-83f7-fcb862011ed3","resolution":{"observed_at":"2026-08-06T17:49:34.760649Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:38.904588Z","title":null,"venue":null,"work_id":"593d96d0-894b-4a58-a81b-a3e14233e415","year":2025},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.764444Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:9e86e5fa4b280ad2d857bc0322c3ecc395949307c766655f4426b3a657dd9c26","observation_id":"6a1f1955-fb2e-4578-932a-2a72e29473a0","resolution":{"observed_at":"2026-08-06T17:49:38.973447Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:34.768376Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.768376Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:dd5e8608bcb6bdbcdfa6f98fa4f397657c4d13867346bb3450cbb0454e9defa2","observation_id":"cdb2164e-c249-4acf-8c87-f125a9f45085","resolution":{"observed_at":"2026-08-06T17:49:34.768376Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:38.750542Z","title":null,"venue":null,"work_id":"994cea36-9b09-40e4-9c58-f8072bdde726","year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.771662Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:fb5a965439da28bdc6e390dfc2932c903af14d09bb98f31d7051592163d2f5c6","observation_id":"1c9ba49d-6aa0-4872-a967-faf832a01226","resolution":{"observed_at":"2026-08-06T17:49:38.817718Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:38.563468Z","title":null,"venue":null,"work_id":"6f94576b-3e2e-4606-ac54-bbfbad1c2fb9","year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.775143Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:ed35035aee0bc3f1eb219434bdcbfabc47ca621cc0515453e5dc296fb2365860","observation_id":"e2612be1-665c-4f42-aef0-71123dcd7b7c","resolution":{"observed_at":"2026-08-06T17:49:38.675109Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:38.409474Z","title":null,"venue":null,"work_id":"1a670a72-f8a9-4260-951c-e0fc4480028c","year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.778450Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:477dfe8cea52c55c3d6cd8a74bdea3157540fde94a9f58c1a99fd226df22c4a9","observation_id":"fc9b25a2-4771-49c2-9ba0-ada62172607b","resolution":{"observed_at":"2026-08-06T17:49:38.456307Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:38.222112Z","title":null,"venue":null,"work_id":"1fd3f72e-f343-4cbd-8854-a51a9187a0ce","year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.781657Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:8cefab29b9a3835797df544124d8642066f15d4e1c547fb90d21139f7034033a","observation_id":"95d610e5-de64-4d0d-9cfd-0afd5792bf15","resolution":{"observed_at":"2026-08-06T17:49:38.318985Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.21776","last_updated":"2025-10-22T16:42:24Z","snapshot_observed_at":"2026-08-05T07:15:29.998948Z","submitted_at":"2025-03-27T17:59:51Z","title":"Video-R1: Reinforcing Video Reasoning in MLLMs","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.21776","snapshot_observed_at":"2026-08-06T17:49:34.784841Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.784841Z"},"links":{"cited_paper":"/paper/2503.21776","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:80c02e99d6c167b76b25fa0bb1b595a93ed1dab850dec67f6b5466c4379d0088","observation_id":"47f20178-e7b6-4f49-b8e5-6375af45dfaf","resolution":{"observed_at":"2026-08-06T17:49:34.784841Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:38.055256Z","title":null,"venue":null,"work_id":"54ac5287-742b-457e-a8d9-f575278ec290","year":2025},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.788646Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:c94ab3ee3f874c0fef98240991c844f608f792a5275b9d5ca54ccc185e274486","observation_id":"40a9f754-3d02-443c-a189-4b4386002a73","resolution":{"observed_at":"2026-08-06T17:49:38.144409Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.13903","last_updated":"2023-11-09T06:50:26Z","snapshot_observed_at":"2026-07-06T15:31:18.144952Z","submitted_at":"2023-05-23T10:26:42Z","title":"Let's Think Frame by Frame with VIP: A Video Infilling and Prediction Dataset for Evaluating Video Chain-of-Thought","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.13903","snapshot_observed_at":"2026-08-06T17:49:34.791982Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.791982Z"},"links":{"cited_paper":"/paper/2305.13903","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:61339132c951b15d2f575d56818992269a2909319d61e9a782f6e5bc4e74632a","observation_id":"34ee85ce-7b5b-4e31-ad35-2200692b9b20","resolution":{"observed_at":"2026-08-06T17:49:34.791982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06428","last_updated":"2025-02-11T14:59:25Z","snapshot_observed_at":"2026-08-08T15:25:00.286848Z","submitted_at":"2025-02-10T13:03:05Z","title":"CoS: Chain-of-Shot Prompting for Long Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.06428","snapshot_observed_at":"2026-08-06T17:49:34.796119Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.796119Z"},"links":{"cited_paper":"/paper/2502.06428","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:59a0f27e7feb71ccfab4b3949e57396bacdd43f7b0f2072c13b46f8a69d54446","observation_id":"2608cca2-ea4e-443e-af2f-9c56d48093d5","resolution":{"observed_at":"2026-08-06T17:49:34.796119Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:37.875522Z","title":null,"venue":null,"work_id":"9516e826-7298-4780-a9a8-622c87748f15","year":2023},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.800952Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:8791d83869af105fe1a574852f5b4e1db061176eb2521643cd7a82750d1de64e","observation_id":"c83bbcbc-7685-4a7b-a003-7427cf56afec","resolution":{"observed_at":"2026-08-06T17:49:37.973782Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:37.688204Z","title":null,"venue":null,"work_id":"454b30bb-bf69-44eb-a089-c53530aaebea","year":2009},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.804448Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:cdf50cc272bc325d0be1cf3f9d233686fec74be6523b2359fc02207b2f36637f","observation_id":"8a6a185a-9cf7-4d98-8d98-077db591bbc2","resolution":{"observed_at":"2026-08-06T17:49:37.797446Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:37.537544Z","title":null,"venue":null,"work_id":"000bba05-3748-4efa-9fa0-289ed7a5ba54","year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.808242Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:2a74e319f1b17be5910e78f5a8772dbd4f9e1c99191b2573322dbf976711a9ea","observation_id":"a2931641-5298-47e4-9ec9-3fe25df594a7","resolution":{"observed_at":"2026-08-06T17:49:37.614690Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:34.812074Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.812074Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:34033753019e13abaee7c64ea0dbae591024f3063a76bc98a7e885d44c11663b","observation_id":"cf46c7f1-1d0f-44e2-aeb7-1e2d646f3f9a","resolution":{"observed_at":"2026-08-06T17:49:34.812074Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1802.03426","last_updated":"2020-09-18T01:56:41Z","snapshot_observed_at":"2026-08-02T15:32:07.466568Z","submitted_at":"2018-02-09T19:39:33Z","title":"UMAP: Uniform Manifold Approximation and Projection for Dimension Reduction","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1802.03426","snapshot_observed_at":"2026-08-06T17:49:34.816549Z","title":null,"venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.816549Z"},"links":{"cited_paper":"/paper/1802.03426","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:7e533740ecc7945eba82083297360ccefd130cc17ee2ed5900abca8eebf32285","observation_id":"705c3741-17e4-41b5-bba2-d05810d930e0","resolution":{"observed_at":"2026-08-06T17:49:34.816549Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:37.361000Z","title":null,"venue":null,"work_id":"cd9a4de3-c38c-4950-835b-cd7b5509b207","year":null},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.821127Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:884772fe6879213c46d667a5b67da2b490666d3313c5c1bd19286e544b9f8d65","observation_id":"99ae4b9e-a7b2-4628-8444-243c9f642b85","resolution":{"observed_at":"2026-08-06T17:49:37.432290Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12819","last_updated":"2025-08-25T02:35:43Z","snapshot_observed_at":"2026-07-06T18:17:21.217971Z","submitted_at":"2024-05-21T14:24:01Z","title":"Large Language Models Meet NLP: A Survey","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.12819","snapshot_observed_at":"2026-08-06T17:49:34.825536Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.825536Z"},"links":{"cited_paper":"/paper/2405.12819","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:5b3afdf1d30347f12547c3eb771366b57825253a768c50d4b2249fec1f81dbd7","observation_id":"8f5a3dfb-fe81-4984-a699-8009d92c7ac0","resolution":{"observed_at":"2026-08-06T17:49:34.825536Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:37.178248Z","title":null,"venue":null,"work_id":"da0b000a-2c92-4c67-9a58-31a15aa1ece5","year":2025},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.829913Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:d95cbe3174a0c7c661c1af366688aa164deba2765ba017f9c9fa236e6d2638ab","observation_id":"543ab5dc-eed6-4180-9e03-722476cbd432","resolution":{"observed_at":"2026-08-06T17:49:37.243312Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:34.833545Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.833545Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:e42e2828a3e5b6ac34bc671903f2724e5f8fc212345eee1b115bbcc59b63a5e8","observation_id":"417c138a-f82d-4e25-8533-f5fecc693e0f","resolution":{"observed_at":"2026-08-06T17:49:34.833545Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:34.836951Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.836951Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:6f1961b4e4aac4900492283370d0d45e9eb7c975b9a624486e032abac1d04752","observation_id":"aaf1e309-0123-40fd-9642-f6257d239dee","resolution":{"observed_at":"2026-08-06T17:49:34.836951Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05530","last_updated":"2024-12-16T17:39:39Z","snapshot_observed_at":"2026-07-06T17:41:42.995949Z","submitted_at":"2024-03-08T18:54:20Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05530","snapshot_observed_at":"2026-08-06T17:49:34.840557Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.840557Z"},"links":{"cited_paper":"/paper/2403.05530","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:1e4d5338f9fbc6246f377d36284d7d41fceb840c8f090f9533208f5d6ec0149b","observation_id":"cdb59fb9-855e-4629-98b2-c2e8e7362bd0","resolution":{"observed_at":"2026-08-06T17:49:34.840557Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:36.971843Z","title":null,"venue":null,"work_id":"a763b27b-245d-481c-9267-3d6309ddb2a2","year":2023},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.844247Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:e4755a21f87f8139132f466269a8cabc71803773bad30132e829d9acab87f1eb","observation_id":"0b118f7e-1b0c-483d-bf9e-cf69301697dd","resolution":{"observed_at":"2026-08-06T17:49:37.066785Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.04091","last_updated":"2023-05-26T07:06:48Z","snapshot_observed_at":"2026-07-06T15:24:07.662207Z","submitted_at":"2023-05-06T16:34:37Z","title":"Plan-and-Solve Prompting: Improving Zero-Shot Chain-of-Thought Reasoning by Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.04091","snapshot_observed_at":"2026-08-06T17:49:34.848049Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.848049Z"},"links":{"cited_paper":"/paper/2305.04091","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:c5b83f6e0256d8f9a1c944b51814fb76b49f72f6fee55f90db1651d81b41676b","observation_id":"c8b75389-fff8-4870-8dad-a03a8435da70","resolution":{"observed_at":"2026-08-06T17:49:34.848049Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:36.820936Z","title":null,"venue":null,"work_id":"21b6cdfc-a337-49bb-a379-71aa7fdc5ad6","year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.852024Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:9c1923201ffc66e363ad0c9bc611ed9af54d52b6c13ab64b5455653d12c475e4","observation_id":"d4367e60-f478-420a-ac14-fdb8ded6754f","resolution":{"observed_at":"2026-08-06T17:49:36.875213Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.06020","last_updated":"2025-09-05T16:04:23Z","snapshot_observed_at":"2026-08-07T15:48:28.204859Z","submitted_at":"2025-05-09T13:08:27Z","title":"ArtRAG: Retrieval-Augmented Generation with Structured Context for Visual Art Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.06020","snapshot_observed_at":"2026-08-06T17:49:34.856382Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.856382Z"},"links":{"cited_paper":"/paper/2505.06020","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:522b6883c4357c62d35a0a4e0ed237a4ff3ea8eb6ec21a7ddf034b05d148265d","observation_id":"e94e0bfa-8ec6-4cfa-aa30-3dd8f9994c8e","resolution":{"observed_at":"2026-08-06T17:49:34.856382Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.11171","last_updated":"2023-03-07T17:57:37Z","snapshot_observed_at":"2026-07-06T12:50:22.773056Z","submitted_at":"2022-03-21T17:48:52Z","title":"Self-Consistency Improves Chain of Thought Reasoning in Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.11171","snapshot_observed_at":"2026-08-06T17:49:34.861248Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.861248Z"},"links":{"cited_paper":"/paper/2203.11171","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:bfcfacb7470ae895f5d055505c60fc08e484fc33d395bf0ae96d3567df93d339","observation_id":"96400931-ec18-4727-a50d-ab57e09dac6f","resolution":{"observed_at":"2026-08-06T17:49:34.861248Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.12605","last_updated":"2025-03-23T13:47:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-16T18:39:13Z","title":"Multimodal Chain-of-Thought Reasoning: A Comprehensive Survey","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.12605","snapshot_observed_at":"2026-08-06T17:49:34.865290Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.865290Z"},"links":{"cited_paper":"/paper/2503.12605","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:5d9cdc82267171da7cae0101871b47a8578538925c37e8c2d1feec4b7665cfad","observation_id":"4039a2da-2b2a-458d-952c-ce10cad2ac97","resolution":{"observed_at":"2026-08-06T17:49:34.865290Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:36.692949Z","title":null,"venue":null,"work_id":"3a8d330d-761e-4335-8f28-5cf6c5eeca57","year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.870094Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:651e12c228fbe5e49e0cdf11d0c04f894f3143dc664d48dc7d3fda1ac986a1b5","observation_id":"a7b84ee6-057a-4995-bdb6-9daf7fb89eb0","resolution":{"observed_at":"2026-08-06T17:49:36.762340Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:36.554713Z","title":null,"venue":null,"work_id":"ce7939ee-2133-4eee-98f9-d2fc25076126","year":2022},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.874308Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:bd1cea4d110b08e720dfa912fc7fe23dcf6cd55cbab6596cb4ae0564b30edce5","observation_id":"07208b62-5b89-4fe8-a7d5-d9468087a54e","resolution":{"observed_at":"2026-08-06T17:49:36.621591Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:34.877874Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.877874Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:f5321ac70bb0b84dd6fdd6d489d503e24928a28808dda768a7b77d3bcd8ddf86","observation_id":"725d5f29-1d0b-4b20-ad2e-647f8f1ac83f","resolution":{"observed_at":"2026-08-06T17:49:34.877874Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:36.271919Z","title":null,"venue":null,"work_id":"402f7410-6513-4bc5-8b2a-62048f70db1c","year":2021},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.885407Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:85b77f8d52ee04da8e88e5324689ea34c39947f4650fe3942e95273ff8cb69db","observation_id":"aa414a16-3ddf-4612-b5aa-f74256674037","resolution":{"observed_at":"2026-08-06T17:49:36.340889Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:36.154008Z","title":null,"venue":null,"work_id":"c1142786-5a6f-400d-b5aa-a7a5fdbc5c3a","year":null},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.889875Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:d5809fa3e12478aa581d049186bd4b58ae0a89343ad6ec593895d9b00e8e3b11","observation_id":"4021bd4b-b315-4079-99d1-370aadc0c1c5","resolution":{"observed_at":"2026-08-06T17:49:36.190327Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.09193","last_updated":"2023-11-15T18:39:21Z","snapshot_observed_at":"2026-08-02T16:45:54.807090Z","submitted_at":"2023-11-15T18:39:21Z","title":"The Role of Chain-of-Thought in Complex Vision-Language Reasoning Task","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.09193","snapshot_observed_at":"2026-08-06T17:49:34.898886Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.898886Z"},"links":{"cited_paper":"/paper/2311.09193","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:a601cbcb61fc73035d7f4607200ed11b9f8c4681583bf9d22c81f7c6f1ad8851","observation_id":"75444443-fc2f-4161-a82e-a565c3d2ca3c","resolution":{"observed_at":"2026-08-06T17:49:34.898886Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:36.032658Z","title":"In Proceedings of the 32nd ACM International Conference on Multimedia","venue":null,"work_id":"3f081fa8-4d61-46b5-9130-9f4d74acf83f","year":null},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.894306Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:8fe274f1d8eab3b791b94c8d600f17a0984826001efc272ca4d5f5ff95eaf1de","observation_id":"81bbbdab-83a1-4ccd-b08e-068c332d6e55","resolution":{"observed_at":"2026-08-06T17:49:36.091285Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:35.945026Z","title":null,"venue":null,"work_id":"4f4db40f-bad7-4244-ba3f-7590a24b0cce","year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.907424Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:7aef819d546636cc6642ab053528c6ee846d67790639284aae8bb7e88085a253","observation_id":"f5c01e62-6511-4ff8-a8b5-38d91ddb0294","resolution":{"observed_at":"2026-08-06T17:49:35.989539Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.15676","last_updated":"2025-02-05T22:01:59Z","snapshot_observed_at":"2026-08-08T23:18:01.787925Z","submitted_at":"2024-04-24T06:12:00Z","title":"Beyond Chain-of-Thought: A Survey of Chain-of-X Paradigms for LLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.15676","snapshot_observed_at":"2026-08-06T17:49:34.903388Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.903388Z"},"links":{"cited_paper":"/paper/2404.15676","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:f2acdb3fe162310cbd27d53e35a2ee454168cabf004409183ec101f1a80faecb","observation_id":"738d0cee-20c3-4b20-9962-7b8c1bdad03c","resolution":{"observed_at":"2026-08-06T17:49:34.903388Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:34.914835Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.914835Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:189b92a5575157dfc2747199f5920e607fa710559c18caa22b08f6e891357d17","observation_id":"de0f54f3-7733-48c3-a7ca-14a2344bf4da","resolution":{"observed_at":"2026-08-06T17:49:34.914835Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:34.911048Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.911048Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:b0c8da72df10641fb196fec4bd4d55c0c2a4386036888b204255fdd612760677","observation_id":"2dce07eb-d539-47f4-b7e2-3ff2e92a9a68","resolution":{"observed_at":"2026-08-06T17:49:34.911048Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:34.922692Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.922692Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:21b3be48a285260fff1dd91327c4e19b13483431c2a536a451b54af387dd3b0f","observation_id":"da72bc3b-1d72-4d21-b00e-4fb3eae1daec","resolution":{"observed_at":"2026-08-06T17:49:34.922692Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-06T17:49:34.919181Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.919181Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:c7091fd12b3173c4828a3d32eefc92e49ddbd4c13ef6952de30e0dd70052e698","observation_id":"fb1ad447-fb3d-417a-bbda-ba92963928ef","resolution":{"observed_at":"2026-08-06T17:49:34.919181Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:35.790666Z","title":null,"venue":null,"work_id":"ef90d9de-ebd3-4d2a-8e7a-2621f889c9f9","year":2025},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.931465Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:98326bd8cbcedf2265147deb9d6ab73a49d0a1f612f432ba1b282864593d440d","observation_id":"c1269154-5f01-4e48-9464-a743702026bd","resolution":{"observed_at":"2026-08-06T17:49:35.848311Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16006","last_updated":"2024-04-24T17:37:05Z","snapshot_observed_at":"2026-07-06T18:05:03.284784Z","submitted_at":"2024-04-24T17:37:05Z","title":"MMT-Bench: A Comprehensive Multimodal Benchmark for Evaluating Large Vision-Language Models Towards Multitask AGI","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16006","snapshot_observed_at":"2026-08-06T17:49:34.927360Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.927360Z"},"links":{"cited_paper":"/paper/2404.16006","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:643697023bd3860cfd553b7735a18d3f1bc2011fa2caf31647948a8b210d39c0","observation_id":"5dfb38a4-4956-47d4-9875-0bea4f982ff9","resolution":{"observed_at":"2026-08-06T17:49:34.927360Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:35.655271Z","title":null,"venue":null,"work_id":"bc6f80c7-2768-4e0e-911d-d84d45495aaf","year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.939634Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:d22a8ad7fdef1c513c0ead008de5f22a024b6d6513baa8f3baf50b325106dbbe","observation_id":"5e3c5ab6-bafd-4e4b-bf04-bdd6ff960e1a","resolution":{"observed_at":"2026-08-06T17:49:35.695185Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13106","last_updated":"2025-06-03T03:33:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T18:59:46Z","title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13106","snapshot_observed_at":"2026-08-06T17:49:34.935033Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.935033Z"},"links":{"cited_paper":"/paper/2501.13106","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:71babb8be27e548a4553710454a7643fa1385fd5fbac586f6d774a3d306f5416","observation_id":"880af460-c720-488d-9016-1775980c506b","resolution":{"observed_at":"2026-08-06T17:49:34.935033Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.18223","last_updated":"2026-03-18T05:34:39Z","snapshot_observed_at":"2026-08-06T23:27:24.356320Z","submitted_at":"2023-03-31T17:28:46Z","title":"A Survey of Large Language Models","version":19},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.18223","snapshot_observed_at":"2026-08-06T17:49:34.948907Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.948907Z"},"links":{"cited_paper":"/paper/2303.18223","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:d1be2e0eba4987307ee80b9e4d5823321e1fa5d80575c9e003acdd3f8b7b0a65","observation_id":"77f6682e-d6f3-4804-97bc-14acf2efeb61","resolution":{"observed_at":"2026-08-06T17:49:34.948907Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19108","last_updated":"2025-05-25T11:54:32Z","snapshot_observed_at":"2026-08-07T14:18:00.295736Z","submitted_at":"2025-05-25T11:54:32Z","title":"CCHall: A Novel Benchmark for Joint Cross-Lingual and Cross-Modal Hallucinations Detection in Large Language Models","version":1},"cited_work":{"arxiv_id":"2505.19108","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.19108","snapshot_observed_at":"2026-08-06T17:49:35.067988Z","title":"CCHall: A Novel Benchmark for Joint Cross-Lingual and Cross-Modal Hallucinations Detection in Large Language Models","venue":"cs.CL","work_id":"30e0f882-481f-431b-9347-64fdecf1b230","year":2025},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.944193Z"},"links":{"cited_paper":"/paper/2505.19108","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:736e154561667776cf1db169838225cfada1eb7745dc87a6fea535affb5639bb","observation_id":"e905d582-7990-4c7a-a245-8611b3d2a1ed","resolution":{"observed_at":"2026-08-06T17:49:35.074557Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04264","last_updated":"2025-01-01T15:53:58Z","snapshot_observed_at":"2026-08-03T20:38:36.602554Z","submitted_at":"2024-06-06T17:09:32Z","title":"MLVU: Benchmarking Multi-task Long Video Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04264","snapshot_observed_at":"2026-08-06T17:49:34.958286Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.958286Z"},"links":{"cited_paper":"/paper/2406.04264","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:a40dbbcbd48a02438d416107bc8f272930d8381570d30bb83386bc995369cc7d","observation_id":"dd4dff17-aeac-4d24-8f70-b2c93b50ac69","resolution":{"observed_at":"2026-08-06T17:49:34.958286Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:34.953969Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.953969Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:b0d31ed31895f3b7bd1c78250153f842006b4967319dea61cfbbd104852ae8ce","observation_id":"5048f173-f12d-46c6-b396-352c5b0f0fb0","resolution":{"observed_at":"2026-08-06T17:49:34.953969Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16038","last_updated":"2024-01-30T14:37:10Z","snapshot_observed_at":"2026-08-06T12:45:18.285554Z","submitted_at":"2024-01-30T14:37:10Z","title":"A Survey on Generative AI and LLM for Video Generation, Understanding, and Streaming","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16038","snapshot_observed_at":"2026-08-06T17:49:34.963022Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.963022Z"},"links":{"cited_paper":"/paper/2404.16038","citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:5740116f817a7f16aa41422a3412ec007340060a65d97ebabc3fe12ee96f8de2","observation_id":"71857a80-a104-4735-aadb-a2a0ed40550d","resolution":{"observed_at":"2026-08-06T17:49:34.963022Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:49:36.426483Z","title":"In European Conference on Computer Vision","venue":null,"work_id":"f92c5565-9054-4640-b2a8-153b8bc451a8","year":null},"citing_paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-06T17:49:34.881390Z"},"links":{"citing_paper":"/paper/2507.09876"},"observation_digest":"sha256:a57e17b36e48360fb7f038243216095c1a2e25ad5a3c1d5a1cec9cf97d9de5ca","observation_id":"88264853-d586-461a-bbb6-a98f2bc7408b","resolution":{"observed_at":"2026-08-06T17:49:36.505134Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.09876","last_updated":"2025-07-14T03:21:13Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-08T07:59:40.732848Z","submitted_at":"2025-07-14T03:21:13Z","title":"ViTCoT: Video-Text Interleaved Chain-of-Thought for Boosting Video Understanding in Large Language Models"},"reference_resolution":{"displayed":59,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":55,"verified_exact":1,"verified_fuzzy":3},"total_outbound_references":59},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 59 of 59 outbound references and 4 inbound Pith citation observations for arXiv:2507.09876."}