{"as_of":"2026-08-10T00:56:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:3f8872bffb08f532a079b07f329915f9b927e27cdf1eca22773a8f911e1d09a1","coverage":[{"denominator":58,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":58,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T15:48:38.262186Z","state":"measured"},{"denominator":59,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":59,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T09:12:09.932226Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.15028","snapshot_observed_at":"2026-08-04T09:12:09.932226Z","title":"Towards video thinking test: A holistic benchmark for advanced video reasoning and understanding","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2510.17045","last_updated":"2026-06-01T07:19:41Z","snapshot_observed_at":"2026-08-04T09:12:02.562884Z","submitted_at":"2025-10-19T23:17:13Z","title":"Video Reasoning without Training","version":2},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-04T09:12:09.932226Z"},"links":{"cited_paper":"/paper/2507.15028","citing_paper":"/paper/2510.17045"},"observation_digest":"sha256:23e021fec914bc6590a5f0dc21489316f9638f170965dfdddcc791db94a3dc83","observation_id":"029bc2f4-af1c-4e0d-9ee6-6c811843f413","resolution":{"observed_at":"2026-08-04T09:12:09.932226Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2507.15028/citation-record","integrity":"/paper/2507.15028/integrity","json":"/paper/2507.15028/citation-record.json","paper":"/paper/2507.15028"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:47.934876Z","title":"Vqa: Visual question answering","venue":null,"work_id":"968a5bce-7a9c-4288-965f-574f83ebec7c","year":2015},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:30.939545Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:12dc681440d1d8f7ff5aadd9041d9af670dae6716a4953269156f881fc8cd565","observation_id":"5df95d99-6457-4549-98e8-bc2296fb4eaf","resolution":{"observed_at":"2026-08-06T15:48:48.049637Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.10818","last_updated":"2024-10-15T17:55:46Z","snapshot_observed_at":"2026-08-09T21:58:00.683460Z","submitted_at":"2024-10-14T17:59:58Z","title":"TemporalBench: Benchmarking Fine-grained Temporal Understanding for Multimodal Video Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.10818","snapshot_observed_at":"2026-08-06T15:48:31.035101Z","title":"Temporalbench: Towards fine-grained temporal understanding for multimodal video models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:31.035101Z"},"links":{"cited_paper":"/paper/2410.10818","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:baee25117ddb1a02f1f2d67c73b25f7a6c11af98f1af619af970ed2a2d62c03e","observation_id":"f91d00e8-c6f6-491f-b1a5-7bdc970597a8","resolution":{"observed_at":"2026-08-06T15:48:31.035101Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:47.679540Z","title":"Collecting highly paral- lel data for paraphrase evaluation","venue":null,"work_id":"48a6013b-eb83-441c-b3af-1a3dedbbf4f2","year":2011},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:31.146121Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:c4a4ba78c00f7053c765f3b7ad30c18df66b8c004939d82fd6b4fca0af00f960","observation_id":"455bff05-8a2e-4bca-bcb1-2426ad781cbc","resolution":{"observed_at":"2026-08-06T15:48:47.771747Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.05271","last_updated":"2025-09-26T12:52:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-06T18:57:08Z","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.05271","snapshot_observed_at":"2026-08-06T15:48:31.309034Z","title":"Expanding performance boundaries of open-source multimodal models with model, data, and test- time scaling","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:31.309034Z"},"links":{"cited_paper":"/paper/2412.05271","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:1a142b49894a9d734e0e1d1ee11abdb3613026908d60971a9661cdace71c0403","observation_id":"5d1d99ad-54ad-465f-83df-4725f87d8d78","resolution":{"observed_at":"2026-08-06T15:48:31.309034Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14432","last_updated":"2025-05-02T16:03:31Z","snapshot_observed_at":"2026-07-06T19:53:56.407834Z","submitted_at":"2024-11-21T18:59:55Z","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.14432","snapshot_observed_at":"2026-08-06T15:48:31.423977Z","title":"Insight-v: Ex- ploring long-chain visual reasoning with multimodal large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:31.423977Z"},"links":{"cited_paper":"/paper/2411.14432","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:3bb556c5a7cbdd85ab18873959ec596922cf4feafc6f5f7ce5353c3a1172ac17","observation_id":"7811dae3-8c3b-4a69-ac98-5634d2a7ae5b","resolution":{"observed_at":"2026-08-06T15:48:31.423977Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.14515","last_updated":"2024-10-30T13:38:10Z","snapshot_observed_at":"2026-07-06T18:34:24.078145Z","submitted_at":"2024-06-20T17:26:01Z","title":"MMBench-Video: A Long-Form Multi-Shot Benchmark for Holistic Video Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.14515","snapshot_observed_at":"2026-08-06T15:48:31.569319Z","title":"Mmbench-video: A long-form multi-shot benchmark for holistic video under- standing","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:31.569319Z"},"links":{"cited_paper":"/paper/2406.14515","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:0ee7b468ddf750287616ed2fd71aeeccc058cc3905dd0fdebe5a3d77e161d856","observation_id":"419a10bf-2fc6-4e6e-a497-af875e6cbe22","resolution":{"observed_at":"2026-08-06T15:48:31.569319Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.21075","last_updated":"2025-05-30T13:08:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-05-31T17:59:47Z","title":"Video-MME: The First-Ever Comprehensive Evaluation Benchmark of Multi-modal LLMs in Video Analysis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.21075","snapshot_observed_at":"2026-08-06T15:48:31.692811Z","title":"Video-mme: The first-ever compre- hensive evaluation benchmark of multi-modal llms in video analysis","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:31.692811Z"},"links":{"cited_paper":"/paper/2405.21075","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:a2ddb2cd57ab4ba96f7651449c547832c66402202778dcb6819935d6872420b9","observation_id":"20cae88d-8699-4879-bab5-a606e491c550","resolution":{"observed_at":"2026-08-06T15:48:31.692811Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:47.480091Z","title":"Agqa: A benchmark for compositional spatio-temporal reasoning","venue":null,"work_id":"c2b6018d-d546-47c4-8a90-c8a9e9dc28d7","year":2021},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:31.798160Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:a9a4e11984c6e957841f442706c2166290874e830a267274fc8f62446e3ebfcb","observation_id":"6601ce9b-8936-4466-a2e9-2e1660cd6698","resolution":{"observed_at":"2026-08-06T15:48:47.570794Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:47.165096Z","title":"Similarity and fea- tures of natural textures","venue":null,"work_id":"8e66a362-bf86-400f-ace1-f974dbe9c716","year":1999},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:31.967725Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:4c57839c5a0d77b6c3ae149ab6c83ef408ff3f19a0a74b23ba85995a010308b6","observation_id":"c5bb15e9-d108-40c0-95af-fc0fd2632a22","resolution":{"observed_at":"2026-08-06T15:48:47.327426Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:46.894405Z","title":"Natural adversarial examples","venue":null,"work_id":"12a70ea2-0d52-4a08-90b0-5a45d8e46dd0","year":2021},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:32.140486Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:e3b7ae3044efbe32b368acb5b0a98627f772f22791c72f4e90286e1d3d5f3c6a","observation_id":"fef48a7e-f210-40f0-a400-b5f99936a1a2","resolution":{"observed_at":"2026-08-06T15:48:47.012478Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:46.672298Z","title":"Video-mmmu: Evaluating knowledge acquisition from multi-discipline pro- fessional videos, 2025","venue":null,"work_id":"8c162250-3475-4007-8048-2a7a2a036942","year":2025},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:32.242853Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:e003dde29f022494a6bfabd6c29b7c3d0424389f0dfcb521835aaff358031aea","observation_id":"db349333-1200-42ed-9571-7d04be0ba812","resolution":{"observed_at":"2026-08-06T15:48:46.740041Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:32.351024Z","title":"Tgif-qa: Toward spatio-temporal reasoning in visual question answering","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:32.351024Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:14e4e96aa62cbc10f9f15917390bbe7bba1d999412694d36f9fd287775ab8ee3","observation_id":"09e9c5cb-554a-4e44-a014-9f9206135d35","resolution":{"observed_at":"2026-08-06T15:48:32.351024Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:46.399398Z","title":"Robust modeling in cognitive science","venue":null,"work_id":"67ab8535-41f0-496c-b77b-1408b46ec1e2","year":2019},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:32.475759Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:0d5555d269cff8c723e7d2f5443779d41672518848cc742c6e6d80ac73c08f2c","observation_id":"ddf0b73b-170f-460c-a1b1-b11ec693ce2a","resolution":{"observed_at":"2026-08-06T15:48:46.500872Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1809.01696","last_updated":"2019-05-07T21:34:05Z","snapshot_observed_at":"2026-07-06T06:59:26.080263Z","submitted_at":"2018-09-05T19:14:11Z","title":"TVQA: Localized, Compositional Video Question Answering","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1809.01696","snapshot_observed_at":"2026-08-06T15:48:32.606188Z","title":"Tvqa: Localized, compositional video question answering","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:32.606188Z"},"links":{"cited_paper":"/paper/1809.01696","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:a296371dd7abd5fc1925eb4d0e6106471b53bd5c5056b8644b57af914bbdb062","observation_id":"3a2e6533-99d6-4ee3-b0d6-3e5259281859","resolution":{"observed_at":"2026-08-06T15:48:32.606188Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:46.139646Z","title":"Mvbench: A comprehensive multi- modal video understanding benchmark, 2023","venue":null,"work_id":"1b6d864c-04d9-46fc-84cf-17e8df53e309","year":2023},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:32.768765Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:e4ea848f7c7eb7ef4bb94e149c67bc1df8067bc2559a5fd445e2a26b7221ff59","observation_id":"1ac2132a-7884-43c0-b223-c3ae4ba09978","resolution":{"observed_at":"2026-08-06T15:48:46.290809Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:32.871605Z","title":"Grounded language-image pre-training","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:32.871605Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:9bb00589bd48c09ad96e06f79bb67077dcebff19beb0bcc391c5643fd42fe1f2","observation_id":"a15c7e92-a2ef-4d72-87a3-cf47fee2a286","resolution":{"observed_at":"2026-08-06T15:48:32.871605Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00476","last_updated":"2024-06-03T04:13:39Z","snapshot_observed_at":"2026-08-04T21:17:37.211833Z","submitted_at":"2024-03-01T12:02:19Z","title":"TempCompass: Do Video LLMs Really Understand Videos?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00476","snapshot_observed_at":"2026-08-06T15:48:33.000709Z","title":"Tempcom- pass: Do video llms really understand videos?arXiv preprint arXiv:2403.00476, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:33.000709Z"},"links":{"cited_paper":"/paper/2403.00476","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:62d67ea6dd034604a7fe254ece3e2aa95add661efe7cd937102b82a567c1839d","observation_id":"7196bd3a-3cfa-4d87-8e80-86f386fdbdef","resolution":{"observed_at":"2026-08-06T15:48:33.000709Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12961","last_updated":"2025-02-27T06:09:46Z","snapshot_observed_at":"2026-07-06T19:18:17.523057Z","submitted_at":"2024-09-19T17:59:51Z","title":"Oryx MLLM: On-Demand Spatial-Temporal Understanding at Arbitrary Resolution","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12961","snapshot_observed_at":"2026-08-06T15:48:33.180292Z","title":"Oryx mllm: On-demand spatial-temporal understanding at arbitrary resolution","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:33.180292Z"},"links":{"cited_paper":"/paper/2409.12961","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:ffff2f676f3a23935ba1b8080e7fe6e92296df80f859033f7108f5ada58eaaa4","observation_id":"ef22287e-6b45-409f-83b5-5da365626f32","resolution":{"observed_at":"2026-08-06T15:48:33.180292Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12966","last_updated":"2024-03-21T16:26:44Z","snapshot_observed_at":"2026-07-06T17:47:13.814205Z","submitted_at":"2024-03-19T17:59:52Z","title":"Chain-of-Spot: Interactive Reasoning Improves Large Vision-Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12966","snapshot_observed_at":"2026-08-06T15:48:33.294475Z","title":"Chain-of-spot: Interactive reasoning improves large vision-language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:33.294475Z"},"links":{"cited_paper":"/paper/2403.12966","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:68062c653a1a826b180f5572bed4716c1ae8bb8943e3125eac8743bb3913bf76","observation_id":"c23e29bd-49d4-4ee2-a256-716b3e5727d9","resolution":{"observed_at":"2026-08-06T15:48:33.294475Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.04328","last_updated":"2025-06-02T19:33:24Z","snapshot_observed_at":"2026-08-09T08:33:21.749172Z","submitted_at":"2025-02-06T18:59:55Z","title":"Ola: Pushing the Frontiers of Omni-Modal Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.04328","snapshot_observed_at":"2026-08-06T15:48:33.446078Z","title":"Ola: Pushing the frontiers of omni-modal language model with progressive modality alignment","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:33.446078Z"},"links":{"cited_paper":"/paper/2502.04328","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:7cdfd2c1d80470d95a5331370c9d809c7f04b55210dac1ab0ee743c80e014800","observation_id":"37c6c68c-dfa0-458c-aa6f-d438fb959ff9","resolution":{"observed_at":"2026-08-06T15:48:33.446078Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:45.886243Z","title":"Video detail caption, 2024","venue":null,"work_id":"143ea81d-f66b-4324-8262-4d86298f4f5e","year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:33.588967Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:d0c20b247bc9154b98b88c7189edcab015dd3a2939d33b675bbc9c93c210edd5","observation_id":"765b1288-3e8a-46f3-a236-2e205096e505","resolution":{"observed_at":"2026-08-06T15:48:46.002685Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:45.675868Z","title":"Video-chatgpt: Towards detailed video understanding via large vision and language models","venue":null,"work_id":"fd2b8a8b-b14f-4ec7-ba3a-fa78c7be67ee","year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:33.712835Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:81565a7f57c74b9d8b16c431e32ab5960c153837239d6e522777f730930d11c5","observation_id":"d89edf69-5883-4e24-b7c9-6462a3481db4","resolution":{"observed_at":"2026-08-06T15:48:45.768516Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:45.433037Z","title":"Egoschema: A diagnostic benchmark for very long- form video language understanding","venue":null,"work_id":"6ff2c6c7-a62a-471a-97e1-325c110eaf81","year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:33.843680Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:88c5f848790e090f6885bf55a9690d89e5fe677a3b5f257cca17388706b87eba","observation_id":"a8110d16-6f0f-469c-847f-5dddec64fb52","resolution":{"observed_at":"2026-08-06T15:48:45.555971Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:45.155189Z","title":"Identifying the perceptual dimensions of visual complexity of scenes","venue":null,"work_id":"9d4b4eb4-0089-4f74-9ae2-2ea3949808e8","year":2004},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:33.963318Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:de8de4b454e507eb500005b3c7b0f5bde146316081d66da23e35100391a8cb73","observation_id":"e2f1aa5c-f826-4ed0-bcbd-fdbc2fef3663","resolution":{"observed_at":"2026-08-06T15:48:45.268685Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:44.843980Z","title":"Hello gpt-4o","venue":null,"work_id":"58f93450-5d5c-413e-af85-9991954f50a6","year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:34.090197Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:3e328e3dcde913e9ec2235e570d947b478a6e9b26984525aad5904adc8464d17","observation_id":"48f1ad09-e144-43d9-8af3-4afa5c535bfe","resolution":{"observed_at":"2026-08-06T15:48:44.992721Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:44.552561Z","title":"Robustness analysis of video- language models against visual and language perturbations","venue":null,"work_id":"4809e0df-b226-4e37-841f-a645d3b65212","year":2022},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:34.215109Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:2c4fb8cf96c0bd01f875b2e3881314c087dc4881ae89cf9ef6e22f7d2a8853c9","observation_id":"8b509cc9-209d-490f-bf22-4e17d8a20f18","resolution":{"observed_at":"2026-08-06T15:48:44.690772Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:44.332717Z","title":"Visual cot: Advancing multi-modal language models with a com- prehensive dataset and benchmark for chain-of-thought rea- soning","venue":null,"work_id":"a5bfdfc1-0b2c-4d15-81bd-63609b355e41","year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:34.327788Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:cfe3f5a6487091c3136b4708bb55a4efec4b86451d20a5c19e0f15c703f5bd58","observation_id":"8624ad6a-af68-4a15-b742-e3fedcfbcd4c","resolution":{"observed_at":"2026-08-06T15:48:44.432636Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:44.107487Z","title":"Complex narratives","venue":null,"work_id":"802da6c0-7adc-4164-afca-a268675e3240","year":2014},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:34.436761Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:267356d94b0f71a37e9b45febed46a4fdad27070679a15b0134317317f9761e8","observation_id":"3c072a10-ee74-403c-baa0-46613aa21e3a","resolution":{"observed_at":"2026-08-06T15:48:44.211901Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:43.871086Z","title":"A standardized set of 260 pictures: norms for name agreement, image agree- ment, familiarity, and visual complexity","venue":null,"work_id":"7fac0123-6001-4c81-a4b7-a1e9564a4663","year":null},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:34.545309Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:68832aac084f2a74a5635a773e3cccd001e6b61ad7cb9185a131bdb7c3c42333","observation_id":"8b651d54-b4a8-49bb-baa6-c740cfbb7d51","resolution":{"observed_at":"2026-08-06T15:48:43.994154Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.08862","last_updated":"2025-02-10T02:00:10Z","snapshot_observed_at":"2026-08-07T23:55:21.811784Z","submitted_at":"2024-08-16T17:44:02Z","title":"Visual Agents as Fast and Slow Thinkers","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.08862","snapshot_observed_at":"2026-08-06T15:48:34.682053Z","title":"Visual agents as fast and slow thinkers","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:34.682053Z"},"links":{"cited_paper":"/paper/2408.08862","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:e8cb4f23b3c5b8be7ec0fe1f0675d32b19a689ef4be5c82090ab48275796e794","observation_id":"83a81e53-1273-43de-940c-88950fadb301","resolution":{"observed_at":"2026-08-06T15:48:34.682053Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:43.615916Z","title":"Curious objects: How vi- sual complexity guides attention and engagement","venue":null,"work_id":"ad1a61fa-e208-4d0c-98e8-a219710b7b8b","year":2021},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:34.828072Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:fea3c02893dbe79da2b804543b2514846ed13358ab082ac0838201284c31fac5","observation_id":"4be120cf-78ac-448b-a092-a75e3fc7285c","resolution":{"observed_at":"2026-08-06T15:48:43.757236Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:43.409073Z","title":"Cognitive load during problem solving: Ef- fects on learning","venue":null,"work_id":"66565297-ed5d-4103-b0c4-c2879f44bcb4","year":1988},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:34.956258Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:19efba34b494e8c26e5a61f7cce7336929de912f10bf0942299e373ab54d2fb2","observation_id":"c6d3932e-6194-486a-946a-42cb352693bb","resolution":{"observed_at":"2026-08-06T15:48:43.510219Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-06T15:48:35.106376Z","title":"Gemini: a family of highly capable multimodal models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:35.106376Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:44c72b657675c3ed48ab627f8b5aeaf73bda5c6ce2e7eccc06da9746472790b8","observation_id":"80026ae7-918d-488a-a749-558296988b73","resolution":{"observed_at":"2026-08-06T15:48:35.106376Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:43.186330Z","title":"Qwen2.5-vl, 2025","venue":null,"work_id":"6622a7a4-c06f-4e35-8063-be20a79dc686","year":2025},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:35.204531Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:9e41e8ac2dba1c5bb30cab9b3fb8181c2a1110238b9335f6220dea8a391e6098","observation_id":"42b5846c-c080-4b1c-b886-c596ca59958e","resolution":{"observed_at":"2026-08-06T15:48:43.298939Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:42.956743Z","title":"Chain-of-thought prompting elicits reasoning in large language models, 2023","venue":null,"work_id":"0de39dbc-3803-4c7d-8d5b-02a3e0f4d7fc","year":2023},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:35.335517Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:1180da0453286754892702554a469af7a1ae19068685db88bf5b3a778f0dbbd3","observation_id":"fc779f9f-e966-485d-b1a5-df5446d54495","resolution":{"observed_at":"2026-08-06T15:48:43.068789Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:42.744992Z","title":"Star: A benchmark for situated reasoning in real-world videos","venue":null,"work_id":"015c6eb0-ca80-4bae-8b7a-d1fc760aa07b","year":2021},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:35.438170Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:a6a09809b2dfb0e2f5e84750e2329ef2927fe7c0c882f25462cdb2d68a4621c6","observation_id":"544bbc8a-a175-4be3-93a3-8ee3c3b255c1","resolution":{"observed_at":"2026-08-06T15:48:42.832074Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:42.437368Z","title":"Longvideobench: A benchmark for long-context interleaved video-language understanding, 2024","venue":null,"work_id":"3e2a4b94-d942-4e41-aedf-e6341a8f1e53","year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:35.549580Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:1c9aa77247cd38c3b856f295809fd044aa80d874f196a0ada75e2a4a6f827f59","observation_id":"d1b49f1d-6949-4034-b8e3-6b30f8aca8ca","resolution":{"observed_at":"2026-08-06T15:48:42.558730Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:42.216183Z","title":"Next-qa: Next phase of question-answering to explaining temporal actions","venue":null,"work_id":"b6fbeae4-3c78-43fb-8ede-63258147079b","year":2021},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:35.677619Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:8b2deeec31b489bfc6b6933c166466e442939e1d8a149018a9620296e19ea46e","observation_id":"b79f48c1-7e3f-451e-8075-04db7d05888d","resolution":{"observed_at":"2026-08-06T15:48:42.321877Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14899","last_updated":"2024-03-22T13:24:35Z","snapshot_observed_at":"2026-07-31T06:40:35.464819Z","submitted_at":"2023-06-26T17:59:55Z","title":"FunQA: Towards Surprising Video Comprehension","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.14899","snapshot_observed_at":"2026-08-06T15:48:35.898141Z","title":"Funqa: Towards surprising video comprehension","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:35.898141Z"},"links":{"cited_paper":"/paper/2306.14899","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:abd4d269f989f1099f984d33ca8798bfba0d5beadffb7b77118dba7f2a711140","observation_id":"be4cf59e-3032-4c5d-bc8b-fb4d9ffe5508","resolution":{"observed_at":"2026-08-06T15:48:35.898141Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:41.908306Z","title":"Video question answer- ing via gradually refined attention over appearance and mo- tion","venue":null,"work_id":"dadc6706-3013-49a2-9090-ff9675cf5b23","year":2017},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:36.010706Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:51ded2f0013ef47416bc57a5b43399e4ab1d698e8dd26b62b6101cba6c8d8f1a","observation_id":"8615b8f0-4b17-4cfd-8b16-7733e1e6d649","resolution":{"observed_at":"2026-08-06T15:48:42.030420Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:41.636169Z","title":"Sutd-trafficqa: A question answering benchmark and an efficient network for video rea- soning over traffic events","venue":null,"work_id":"9c4dec88-49a3-4418-b4aa-b518d7a37b3b","year":2021},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:36.149550Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:e80d73590f4076c86975f191970a358cda0b80059545c35122fab0968353a2d9","observation_id":"e765a4ed-3fc3-4c43-a5a5-09eb60da7363","resolution":{"observed_at":"2026-08-06T15:48:41.727394Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1910.01442","last_updated":"2020-03-08T00:09:07Z","snapshot_observed_at":"2026-07-06T08:26:38.349660Z","submitted_at":"2019-10-03T13:16:36Z","title":"CLEVRER: CoLlision Events for Video REpresentation and Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1910.01442","snapshot_observed_at":"2026-08-06T15:48:36.257872Z","title":"Clevrer: Collision events for video representation and reasoning","venue":null,"work_id":null,"year":1910},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:36.257872Z"},"links":{"cited_paper":"/paper/1910.01442","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:929ec4a0e2cdb83cd6b63e3be37dba66aa6fb72aba6fc15cff78fe40dae848e7","observation_id":"698ac212-0b18-4618-aebe-1225e15ab204","resolution":{"observed_at":"2026-08-06T15:48:36.257872Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:41.418479Z","title":"Activitynet-qa: A dataset for understanding complex web videos via question answering","venue":null,"work_id":"1ec97e30-39e9-4d29-bb5b-e84dfe02f0f7","year":2019},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:36.361773Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:4801f2025222c6d42126a9940e342acac81912bd568d28fdad32f10d4cfdbf49","observation_id":"b02e17e6-0383-4c89-b9c7-9ef1a00d10c9","resolution":{"observed_at":"2026-08-06T15:48:41.517617Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:41.162839Z","title":"Activitynet-qa: A dataset for understanding complex web videos via question answering","venue":null,"work_id":"663a5957-adad-4089-8885-a14c2efea13d","year":2019},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:36.499719Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:07517a39934467f338e18a65fe93791eba920e6925804d39cc2f3b4839396865","observation_id":"b49b54d2-abcb-4a86-aee5-8c7353405cd7","resolution":{"observed_at":"2026-08-06T15:48:41.286050Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:40.893094Z","title":"Social-iq: A question answer- ing benchmark for artificial social intelligence","venue":null,"work_id":"1af5b41b-ae24-4c93-9334-cea56cab7464","year":2019},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:36.656128Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:4a3356e7522bd1e9c2163da271e1e122f75348264b1bff17af3c376e5c909be3","observation_id":"29c5a716-8008-4753-9afe-d0c154f4e367","resolution":{"observed_at":"2026-08-06T15:48:41.022509Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.02858","last_updated":"2023-10-25T06:23:31Z","snapshot_observed_at":"2026-07-06T15:38:39.712379Z","submitted_at":"2023-06-05T13:17:27Z","title":"Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.02858","snapshot_observed_at":"2026-08-06T15:48:36.756457Z","title":"Video-llama: An instruction-tuned audio-visual language model for video un- derstanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:36.756457Z"},"links":{"cited_paper":"/paper/2306.02858","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:561ff90cf545c727e3f43cd4c25b2b8fb173f8e09d7c81edee154b9611053411","observation_id":"263fd12e-d9a0-49d4-8027-6217a5f54e3e","resolution":{"observed_at":"2026-08-06T15:48:36.756457Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:40.621301Z","title":"B- avibench: Towards evaluating the robustness of large vision- language model on black-box adversarial visual-instructions,","venue":null,"work_id":"225b70f2-3394-4894-a640-91fb863f1529","year":null},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:36.911831Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:434069230326bf468d2f527fc9e8bd4441b097efc64f9c1a9958f794de826477","observation_id":"bbe8c10f-0b64-438a-b966-196f252f44a8","resolution":{"observed_at":"2026-08-06T15:48:40.773699Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:37.032657Z","title":"Lmms- eval: Reality check on the evaluation of large multimodal models, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:37.032657Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:b0a1c255433b8709d5b7750112775f8556c961468033217d062ff6e896ae3e5a","observation_id":"7d7965d5-2b29-4fac-8610-7ec7b3bbff7a","resolution":{"observed_at":"2026-08-06T15:48:37.032657Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.16852","last_updated":"2024-07-01T02:59:29Z","snapshot_observed_at":"2026-08-07T09:52:45.942315Z","submitted_at":"2024-06-24T17:58:06Z","title":"Long Context Transfer from Language to Vision","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.16852","snapshot_observed_at":"2026-08-06T15:48:37.124321Z","title":"Long context transfer from language to vision","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:37.124321Z"},"links":{"cited_paper":"/paper/2406.16852","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:f466e3ab15dd4d75cdcee76c8a7fe7fd167cabf05ad7a425191b2ecd0adf8e2c","observation_id":"fe22ba31-6ef0-422e-a89e-4b5188af112e","resolution":{"observed_at":"2026-08-06T15:48:37.124321Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:40.352479Z","title":"Llava- next: A strong zero-shot video understanding model, 2024","venue":null,"work_id":"f9273919-402d-4881-aa20-2386964149b7","year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:37.232691Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:92b47b4e8ae8e23062f925e35192c31d32feade78c0299180ce7642c251c63ac","observation_id":"0d4b1904-9a86-4848-a923-e83820d76d7e","resolution":{"observed_at":"2026-08-06T15:48:40.466201Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:40.108946Z","title":"Video instruction tuning with synthetic data, 2024","venue":null,"work_id":"582c093d-8b28-4262-8dd6-6f2982998f7a","year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:37.388074Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:8d7a4e6bbf64bf7e2ff833136a748851f734e48fba8b0619a836cb105ead5a51","observation_id":"8364d8e5-e494-420c-88ef-7445affc3200","resolution":{"observed_at":"2026-08-06T15:48:40.216826Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:39.857946Z","title":"Worldqa: Multimodal world knowledge in videos through long-chain reasoning, 2024","venue":null,"work_id":"41ee5d78-ad52-438e-92ed-0c4c3b2f2299","year":2024},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:37.519206Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:b4c42b11a551818705533c5bbf8ecabd4844d3ea33af22de359b8031f99b6252","observation_id":"91e9c0ea-54b4-49e1-91ac-05896924ac87","resolution":{"observed_at":"2026-08-06T15:48:39.995188Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04264","last_updated":"2025-01-01T15:53:58Z","snapshot_observed_at":"2026-08-03T20:38:36.602554Z","submitted_at":"2024-06-06T17:09:32Z","title":"MLVU: Benchmarking Multi-task Long Video Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04264","snapshot_observed_at":"2026-08-06T15:48:37.637372Z","title":"Mlvu: A comprehensive benchmark for multi-task long video understanding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:37.637372Z"},"links":{"cited_paper":"/paper/2406.04264","citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:b5e1717ac22c33aaf19b1b41852bee6c087fb2af67512411fc7a346b3be9a4b6","observation_id":"2571e13e-793f-4fc5-bb6d-3ebd87c5ccdd","resolution":{"observed_at":"2026-08-06T15:48:37.637372Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:39.645273Z","title":"Hierarchical video content description and summarization using unified semantic and visual similarity","venue":null,"work_id":"d6457f4f-efae-442a-bc6a-3638934b16cf","year":2003},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:37.755434Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:0a47ecd9102dde3249d8181e7dfc51de1f697079509b1ed4328888b08118aa63","observation_id":"ddccc800-85a9-4126-838a-ac6ec14d2461","resolution":{"observed_at":"2026-08-06T15:48:39.748619Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:39.331297Z","title":"In total, the annotation process cost 8227.32 human hours","venue":null,"work_id":"88293fa2-49fc-4913-a82b-bc64bf2e33ba","year":null},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:37.878023Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:94d87249b597280d0209daa38a01d5c6457493a785999870bf435c259386ceb6","observation_id":"12f08e3e-6f6d-4d24-8a18-9afa369badf2","resolution":{"observed_at":"2026-08-06T15:48:39.494867Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:39.081973Z","title":"• Aparaphrased correct be the set of videos where the para- phrased open-ended question is answered correctly","venue":null,"work_id":"70b7dde5-dcf9-42cd-a411-9a78d35ccd59","year":null},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:37.984186Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:2596670d2220b4e3087f0223ffc6dab171ba9c531c5c7ab0ef982318f6716083","observation_id":"5e1f3085-8a05-494a-8fe5-7f5260b953d0","resolution":{"observed_at":"2026-08-06T15:48:39.209624Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:38.862420Z","title":"3 shows the prompt for evaluating open-ended an- swers","venue":null,"work_id":"71f834d3-6d97-49fd-9599-db830df97928","year":null},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:38.104212Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:a3413e523731ee4af0fa725d316c0af1607b44f50c152696d7f48f53336af302","observation_id":"4661c413-e575-40cf-8cb5-b6b4260aa7d2","resolution":{"observed_at":"2026-08-06T15:48:38.958699Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:48:38.649973Z","title":"element” and “event","venue":null,"work_id":"cdbe8f52-3a99-49a9-acfa-3d4fbb222535","year":null},"citing_paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:38.262186Z"},"links":{"citing_paper":"/paper/2507.15028"},"observation_digest":"sha256:b35b0aeeec8ea5bc45ccf9696466f4e3e42bed1b3c7a52e657ae615db2614970","observation_id":"f2b4833d-0cf6-4494-a4ee-ad5b033b6135","resolution":{"observed_at":"2026-08-06T15:48:38.751950Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.15028","last_updated":"2025-07-20T16:30:33Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-09T21:07:25.586579Z","submitted_at":"2025-07-20T16:30:33Z","title":"Towards Video Thinking Test: A Holistic Benchmark for Advanced Video Reasoning and Understanding"},"reference_resolution":{"displayed":58,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":20,"verified_exact":0,"verified_fuzzy":38},"total_outbound_references":58},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 58 of 58 outbound references and 1 inbound Pith citation observation for arXiv:2507.15028."}