{"as_of":"2026-08-09T23:22:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:74c06eb9ee18637faf06096b7980011caf017b1eb4d8ae6c99e1ed5bffa505c3","coverage":[{"denominator":44,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":44,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-01T21:09:45.965576Z","state":"measured"},{"denominator":44,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":44,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2607.16189/citation-record","integrity":"/paper/2607.16189/integrity","json":"/paper/2607.16189/citation-record.json","paper":"/paper/2607.16189"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2511.21631","last_updated":"2025-11-27T12:16:54Z","snapshot_observed_at":"2026-07-06T22:37:03.716474Z","submitted_at":"2025-11-26T17:59:08Z","title":"Qwen3-VL Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2511.21631","snapshot_observed_at":"2026-08-01T21:09:41.763169Z","title":"Qwen3-vl technical report.arXiv preprint arXiv:2511.21631, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:41.763169Z"},"links":{"cited_paper":"/paper/2511.21631","citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:9b2d7ca82241cb58744bcab69a79710b80c458bc3db70aff4481c39d8dda19d0","observation_id":"0bb08b37-1d7a-4026-bb59-b6966e225fc1","resolution":{"observed_at":"2026-08-01T21:09:41.763169Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T21:09:41.875217Z","title":"Revisiting the “video” in video-language understanding","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:41.875217Z"},"links":{"citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:df905341f916496b646c4be7dbb023d26ccfe1be7df281eb275232a663260ce7","observation_id":"2cd76cc4-84a5-4b04-a048-2c579d8112cf","resolution":{"observed_at":"2026-08-01T21:09:41.875217Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T21:09:41.954161Z","title":"VideoMiner: Iteratively grounding key frames of hour-long videos via tree-based group relative policy optimization.arXiv preprint arXiv:2510.06040, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:41.954161Z"},"links":{"citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:bebfc5cabb59d68b09425d312d7d9922dcedff3faaac9ca33d72465bde7e9933","observation_id":"fbaa2b13-bfd2-4873-bcee-8ca162105cd8","resolution":{"observed_at":"2026-08-01T21:09:41.954161Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T21:09:42.030375Z","title":"CG-Bench: Clue-grounded question answering benchmark for long video understanding","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:42.030375Z"},"links":{"citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:a3907ca00d41c202cec7a70148933b81aec2d33e402649012cd1b7f8ada45d1a","observation_id":"2e9ccef5-c63d-42d8-94b8-88ec37652e5b","resolution":{"observed_at":"2026-08-01T21:09:42.030375Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T21:09:42.195155Z","title":"ShareGPT4Video: Improving video understanding and generation with better captions.Advances in Neural Information Processing Systems (NeurIPS), 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:42.195155Z"},"links":{"citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:d702a76fe9b5a66e1a03f07ccdee6ec92949417e83e67cbcfc35f8044a38415a","observation_id":"44046130-c663-4c9c-98a3-c56a08721efa","resolution":{"observed_at":"2026-08-01T21:09:42.195155Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.10188","last_updated":"2024-12-13T02:32:06Z","snapshot_observed_at":"2026-08-05T14:57:53.592979Z","submitted_at":"2024-08-19T17:48:08Z","title":"LongVILA: Scaling Long-Context Visual Language Models for Long Videos","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.10188","snapshot_observed_at":"2026-08-01T21:09:42.330014Z","title":"LongVILA: Scaling long-context visual language models for long videos.arXiv preprint arXiv:2408.10188, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:42.330014Z"},"links":{"cited_paper":"/paper/2408.10188","citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:d82133ecd3c2d434c2c5ca9155e7c71c529d5267d66426925ece5d602259734c","observation_id":"e31af0a0-7e39-488b-9e68-b986d5e74fa7","resolution":{"observed_at":"2026-08-01T21:09:42.330014Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T21:09:42.446315Z","title":"VideoZoomer: Reinforcement-learned temporal focusing for long video reasoning.arXiv preprint arXiv:2512.22315, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:42.446315Z"},"links":{"citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:006f29d59e2781955a184ee2e69f7ecf63e882d737a872ef93c963637fb43350","observation_id":"43099faf-9860-4b49-99d3-a85d55c1c44f","resolution":{"observed_at":"2026-08-01T21:09:42.446315Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.21776","last_updated":"2025-10-22T16:42:24Z","snapshot_observed_at":"2026-08-05T07:15:29.998948Z","submitted_at":"2025-03-27T17:59:51Z","title":"Video-R1: Reinforcing Video Reasoning in MLLMs","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.21776","snapshot_observed_at":"2026-08-01T21:09:42.508311Z","title":"Video-r1: Reinforcing video reasoning in mllms.arXiv preprint arXiv:2503.21776, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:42.508311Z"},"links":{"cited_paper":"/paper/2503.21776","citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:6062d87a6c0d663bfe6b236ae7ba1dd2652e899b5faa6d7845b53c02c772e5da","observation_id":"d6c9e309-563c-4c60-8957-33fe452cd3ca","resolution":{"observed_at":"2026-08-01T21:09:42.508311Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T21:09:42.622277Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:42.622277Z"},"links":{"citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:8c5ff400d1d7244f3bd9102b1876ffa520b668b47b65f0d8ed065370db710faa","observation_id":"be5e3cd1-0857-4f95-9390-550c22a657e4","resolution":{"observed_at":"2026-08-01T21:09:42.622277Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T21:09:42.728739Z","title":"ReVisionLLM: Recursive vision-language model for temporal grounding in hour-long videos","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:42.728739Z"},"links":{"citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:a9a16d557fbc1ff1467505979d7206306decda7f372ea3d0d2cd435d1f22bb9f","observation_id":"6babdc4c-0950-4397-8730-7ec9fca0dbaf","resolution":{"observed_at":"2026-08-01T21:09:42.728739Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T21:09:42.910965Z","title":"Long movie clip classification with state-space video models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:42.910965Z"},"links":{"citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:98e595a3a0ce852a81f70fd94c948d793d16cf6da52b411ab813a7d907c4e166","observation_id":"df930178-2b3a-43b9-a0b3-2c4480914b84","resolution":{"observed_at":"2026-08-01T21:09:42.910965Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T21:09:43.043540Z","title":"Efficient movie scene detection using state-space transformers","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:43.043540Z"},"links":{"citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:1ed7f61f1fb25b94ec3dccc638b009d5144864036e76f1a5477a15ce7a4e72a2","observation_id":"40fca317-5763-4c9f-bd62-79b95b4e733d","resolution":{"observed_at":"2026-08-01T21:09:43.043540Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T21:09:43.128553Z","title":"Video ReCap: Recursive captioning of hour-long videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:43.128553Z"},"links":{"citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:9996dd388bc307acaa0cc5cad20f70af658470c96a2b91dfa6040fa6073c0029","observation_id":"44921443-2857-4240-83d4-3e84c564af26","resolution":{"observed_at":"2026-08-01T21:09:43.128553Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T21:09:43.287488Z","title":"BIMBA: Selective-scan compression for long-range video question answering","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:43.287488Z"},"links":{"citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:ec788d450ccd3e370b9b0d2e773d0c1afe6ec20d5d21d5b8617a5cf753d5e08e","observation_id":"e3fb1495-cc4e-4279-873e-002b7ba41fa5","resolution":{"observed_at":"2026-08-01T21:09:43.287488Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T21:09:43.382041Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:43.382041Z"},"links":{"citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:8816e18ab85d6983c56c919e6bb67c8e0505ef2ad7a025f55730a7362b880bed","observation_id":"2a9c70b5-7d51-4f0b-a03b-b8e2fc64a5bd","resolution":{"observed_at":"2026-08-01T21:09:43.382041Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.06958","last_updated":"2025-11-11T08:30:00Z","snapshot_observed_at":"2026-08-02T02:31:33.589341Z","submitted_at":"2025-04-09T15:09:27Z","title":"VideoChat-R1: Enhancing Spatio-Temporal Perception via Reinforcement Fine-Tuning","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.06958","snapshot_observed_at":"2026-08-01T21:09:43.463242Z","title":"VideoChat-R1: Enhancing spatio-temporal perception via reinforce- ment fine-tuning.arXiv preprint arXiv:2504.06958, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:43.463242Z"},"links":{"cited_paper":"/paper/2504.06958","citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:36536a54f3a630c7726ff51f0102c5eb0c694bf56aa88f9ceb73a996e84d56c7","observation_id":"2d96c88b-f6e2-46a0-8165-73e347507928","resolution":{"observed_at":"2026-08-01T21:09:43.463242Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T21:09:43.571420Z","title":"LLaMA-VID: An image is worth 2 tokens in large language models.European Conference on Computer Vision (ECCV), 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:43.571420Z"},"links":{"citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:48ec857d5f1ff82d6488d041f725db55fe85a9a4164c44be7836cafcad31161d","observation_id":"ca4c39a9-0a3f-4708-8551-7ace2a98966f","resolution":{"observed_at":"2026-08-01T21:09:43.571420Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T21:09:43.682875Z","title":"TimeSearch-R: Adaptive temporal search for long-form video understanding via self-verification reinforcement learning.arXiv preprint arXiv:2511.05489, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:43.682875Z"},"links":{"citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:c13c0f5d695e8f18ac32b6dd2400962d8204b1f3bacda81270a07f4b44a578a9","observation_id":"bbe511cd-bbb2-40ce-8e78-b68d1ce62936","resolution":{"observed_at":"2026-08-01T21:09:43.682875Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T21:09:43.802551Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:43.802551Z"},"links":{"citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:c9f0b7963d6fb41f7b5868e540fc98d5a13863c6e827ed5938d035cb20e18377","observation_id":"84c09683-d885-4508-b4c4-a0212dd8815e","resolution":{"observed_at":"2026-08-01T21:09:43.802551Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T21:09:43.929348Z","title":"Learning transferable visual models from natural language supervision","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:43.929348Z"},"links":{"citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:320e5cd83cf676e5a5281b2706d0018b0e1c5a0faab2d37ab6fd900f37797202","observation_id":"62e4b502-7a54-4aea-8e2c-6dcf227a67f6","resolution":{"observed_at":"2026-08-01T21:09:43.929348Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T21:09:44.052763Z","title":"Understanding long videos in one multimodal language model pass","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:44.052763Z"},"links":{"citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:09727365046a1da0b2322914dc984aaa0c9ec3cd052636587b94dd4413644245","observation_id":"3f14f7b0-4d50-4382-8e79-5db2ade9af63","resolution":{"observed_at":"2026-08-01T21:09:44.052763Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T21:09:44.148009Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:44.148009Z"},"links":{"citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:5be1986f21e6f861f30ecf4e2bd3a04e0ce91b96f506401fa12dce972830ef6f","observation_id":"fdeeffc3-46f4-4f02-b7e9-6321d1ef67df","resolution":{"observed_at":"2026-08-01T21:09:44.148009Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.16267","last_updated":"2025-06-09T19:33:32Z","snapshot_observed_at":"2026-07-06T19:37:16.203681Z","submitted_at":"2024-10-21T17:59:11Z","title":"xGen-MM-Vid (BLIP-3-Video): You Only Need 32 Tokens to Represent a Video Even in VLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.16267","snapshot_observed_at":"2026-08-01T21:09:44.272531Z","title":"Ryoo, Honglu Zhou, Shrikant Kendre, Can Qin, Le Xue, Manli Shu, Jongwoo Park, Kanchana Ranasinghe, Silvio Savarese, Ran Xu, Caiming Xiong, and Juan Carlos Niebles","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:44.272531Z"},"links":{"cited_paper":"/paper/2410.16267","citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:9226fb4d81cf99d5fe289e6dad8ef612c04fc5de66eb8769a712332182b75019","observation_id":"b7a55875-f353-4b32-9ba8-3b1c1b284223","resolution":{"observed_at":"2026-08-01T21:09:44.272531Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.17434","last_updated":"2024-10-22T21:21:37Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-22T21:21:37Z","title":"LongVU: Spatiotemporal Adaptive Compression for Long Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.17434","snapshot_observed_at":"2026-08-01T21:09:44.385817Z","title":"LongVU: Spa- tiotemporal adaptive compression for long video-language understanding.arXiv preprint arXiv:2410.17434, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:44.385817Z"},"links":{"cited_paper":"/paper/2410.17434","citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:d6911ab7a1da6d411712a8a47b46071b7989fab482ef78e8788ad79b8ba65474","observation_id":"751eb134-49f0-4ee9-953f-35fff8f7cfeb","resolution":{"observed_at":"2026-08-01T21:09:44.385817Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T21:09:44.512577Z","title":"MovieChat: From dense token to sparse memory for long video understanding.Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR), 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:44.512577Z"},"links":{"citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:785d1f2a8df526f1e4ca6f5ec8283b2a4d02851c589793f9aa447806f9be03ed","observation_id":"b5d2c529-0028-4e4a-b1de-c4ff5247ef5c","resolution":{"observed_at":"2026-08-01T21:09:44.512577Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T21:09:44.636461Z","title":"Adaptive keyframe sampling for long video understanding","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:44.636461Z"},"links":{"citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:04b2726ed6899fb4a7cc848c3dcabcf3e3da77179803c10667a5e60e4b13367e","observation_id":"4955792f-f0e2-48f5-b1f6-b3a1c10e2f66","resolution":{"observed_at":"2026-08-01T21:09:44.636461Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T21:09:44.742647Z","title":"Alvarez, Lei Zhang, and Zhiding Yu","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:44.742647Z"},"links":{"citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:34646ce38e706105b54e50721a09516a27c79234c8aae3fe1ca96cbb0a9734cc","observation_id":"4a7e6fa5-5466-454f-8284-109e92a63dce","resolution":{"observed_at":"2026-08-01T21:09:44.742647Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08035","last_updated":"2025-08-09T10:54:59Z","snapshot_observed_at":"2026-08-08T19:38:26.415599Z","submitted_at":"2024-06-12T09:36:52Z","title":"LVBench: An Extreme Long Video Understanding Benchmark","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08035","snapshot_observed_at":"2026-08-01T21:09:44.860609Z","title":"LVBench: An extreme long video understanding benchmark.arXiv preprint arXiv:2406.08035, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:44.860609Z"},"links":{"cited_paper":"/paper/2406.08035","citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:beca28649a2dade35aa44b5023d537d6671d0fa2f90ba417359e94f9db8fe9f5","observation_id":"d06acb4e-218a-443d-ad58-723e803b012f","resolution":{"observed_at":"2026-08-01T21:09:44.860609Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.12559","last_updated":"2025-06-08T05:57:12Z","snapshot_observed_at":"2026-08-07T17:00:27.328884Z","submitted_at":"2025-03-16T16:14:52Z","title":"AdaReTaKe: Adaptive Redundancy Reduction to Perceive Longer for Video-language Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.12559","snapshot_observed_at":"2026-08-01T21:09:44.952695Z","title":"AdaReTaKe: Adaptive redundancy reduction to perceive longer for video-language understanding.arXiv preprint arXiv:2503.12559, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:44.952695Z"},"links":{"cited_paper":"/paper/2503.12559","citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:df978ece841fbad1c5eedb97fc7b76df04079efec30b308d49954bedbfe8a589","observation_id":"659a6dfb-9521-4811-ab31-a4ca266faee6","resolution":{"observed_at":"2026-08-01T21:09:44.952695Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.13377","last_updated":"2025-06-29T08:11:35Z","snapshot_observed_at":"2026-08-02T21:01:10.455745Z","submitted_at":"2025-03-17T17:04:20Z","title":"Time-R1: Post-Training Large Vision Language Model for Temporal Video Grounding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.13377","snapshot_observed_at":"2026-08-01T21:09:45.070624Z","title":"TimeZero: Temporal video grounding with reasoning-guided LVLM.arXiv preprint arXiv:2503.13377, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:45.070624Z"},"links":{"cited_paper":"/paper/2503.13377","citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:30135653295827001b47d7f4f71d8691b623a3d3c8bac3117b0949408f7cce2f","observation_id":"c4169a6b-2f5d-4b69-910a-9db16aedc76f","resolution":{"observed_at":"2026-08-01T21:09:45.070624Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T21:09:45.141034Z","title":"VideoTree: Adaptive tree-based video representation for LLM reasoning on long videos","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:45.141034Z"},"links":{"citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:8fd595855ef66a83165b7955dffdeb87e6de1d43f4d09cae2626976d79dccda7","observation_id":"ca71d312-04bf-4640-9f7a-976d50d0a3c2","resolution":{"observed_at":"2026-08-01T21:09:45.141034Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2512.05774","last_updated":"2026-06-04T07:27:00Z","snapshot_observed_at":"2026-08-03T18:22:22.611333Z","submitted_at":"2025-12-05T15:03:48Z","title":"Active Video Perception: Iterative Evidence Seeking for Agentic Long Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2512.05774","snapshot_observed_at":"2026-08-01T21:09:45.205921Z","title":"Ryoo, and Juan Carlos Niebles","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:45.205921Z"},"links":{"cited_paper":"/paper/2512.05774","citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:023f1d121ba10e86220742dc18a5fb0d777fda4efa549cca5f6f460e6a21135e","observation_id":"b2160f28-107f-4777-aa54-4fcf8fa980b3","resolution":{"observed_at":"2026-08-01T21:09:45.205921Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T21:09:45.254499Z","title":"LongVideoBench: A benchmark for long- context interleaved video-language understanding.Advances in Neural Information Processing Systems (NeurIPS), 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:45.254499Z"},"links":{"citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:6aaf3c3a68619ee37d4807ac3e216592208401bdae9c1d1c34405d2a9f213925","observation_id":"47ada167-aec7-4386-aa2c-5396ba2e10b0","resolution":{"observed_at":"2026-08-01T21:09:45.254499Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2511.20785","last_updated":"2026-05-21T10:39:51Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-11-25T19:22:48Z","title":"LongVT: Incentivizing \"Thinking with Long Videos\" via Native Tool Calling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2511.20785","snapshot_observed_at":"2026-08-01T21:09:45.294808Z","title":"thinking with long videos","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:45.294808Z"},"links":{"cited_paper":"/paper/2511.20785","citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:ae4639a700a70a2739b94ecd6fa99aadce18d039de1eb770b2f2b40ba1e1be81","observation_id":"7ffb45fa-d594-474d-aa85-e13fd2a50c35","resolution":{"observed_at":"2026-08-01T21:09:45.294808Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09146","last_updated":"2025-09-02T09:52:40Z","snapshot_observed_at":"2026-08-07T17:10:48.940585Z","submitted_at":"2025-03-12T08:16:39Z","title":"Generative Frame Sampler for Long Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09146","snapshot_observed_at":"2026-08-01T21:09:45.384742Z","title":"Generative frame sampler for long video understanding.arXiv preprint arXiv:2503.09146, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:45.384742Z"},"links":{"cited_paper":"/paper/2503.09146","citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:7dea550c3ff8b6c23b204862233a06bfea6cda549a9d3e55f65ba62fd426d223","observation_id":"02b1174a-3beb-4864-aa68-2cf561590e83","resolution":{"observed_at":"2026-08-01T21:09:45.384742Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.02259","last_updated":"2025-08-25T02:57:46Z","snapshot_observed_at":"2026-08-07T18:21:41.515675Z","submitted_at":"2025-04-03T04:03:10Z","title":"T*: Re-thinking Temporal Search for Long-Form Video Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.02259","snapshot_observed_at":"2026-08-01T21:09:45.454954Z","title":"Re-thinking temporal search for long-form video understanding.arXiv preprint arXiv:2504.02259, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:45.454954Z"},"links":{"cited_paper":"/paper/2504.02259","citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:ebdca5cd4fa42b852babd408824643b1a52a6048fd0ad28c519733ef132f658e","observation_id":"3bb651f5-d098-4d35-8655-7cfa911c0247","resolution":{"observed_at":"2026-08-01T21:09:45.454954Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T21:09:45.511098Z","title":"MomentSeeker: A benchmark for long-video moment retrieval.arXiv preprint arXiv:2502.12558, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:45.511098Z"},"links":{"citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:8b0994a2b9d2284f9438bcc02fde5f6cc15c16378d64929f3266afa1aa9b4218","observation_id":"c761285d-106e-48eb-a426-a80e90c012f3","resolution":{"observed_at":"2026-08-01T21:09:45.511098Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2601.23224","last_updated":"2026-05-21T09:05:36Z","snapshot_observed_at":"2026-08-02T12:29:31.448544Z","submitted_at":"2026-01-30T17:47:30Z","title":"Video-o3: Native Interleaved Clue Seeking for Long Video Multi-Hop Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2601.23224","snapshot_observed_at":"2026-08-01T21:09:45.593997Z","title":"Video-o3: Native interleaved clue seeking for long video multi-hop reasoning.arXiv preprint arXiv:2601.23224, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:45.593997Z"},"links":{"cited_paper":"/paper/2601.23224","citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:b948c36f0b7826593779cc1d2676fa3e25f2c11c67bef12ad40b06d631d8ff4a","observation_id":"2d04a5fb-0e4a-4430-bed8-93c0705a9293","resolution":{"observed_at":"2026-08-01T21:09:45.593997Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T21:09:45.641212Z","title":"A simple LLM framework for long-range video question-answering","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:45.641212Z"},"links":{"citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:855cc322a19dbc1549ed9f144b09a7b8c7789f9a07b7556d550723cd58b0af05","observation_id":"80010330-3aa0-499f-8133-04338642dd97","resolution":{"observed_at":"2026-08-01T21:09:45.641212Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.24869","last_updated":"2026-04-15T15:09:43Z","snapshot_observed_at":"2026-07-06T21:33:50.814913Z","submitted_at":"2025-05-30T17:59:19Z","title":"SiLVR: A Simple Language-based Video Reasoning Framework","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.24869","snapshot_observed_at":"2026-08-01T21:09:45.691759Z","title":"SiLVR: A simple language-based video reasoning framework.arXiv preprint arXiv:2505.24869, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:45.691759Z"},"links":{"cited_paper":"/paper/2505.24869","citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:a96e7e44fc6323998ec99d15552ea26f0c17c439c1a067485cb5a44516528e21","observation_id":"bc8524ad-ed63-4605-8c9d-a10dec30233e","resolution":{"observed_at":"2026-08-01T21:09:45.691759Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.04416","last_updated":"2025-09-03T07:11:03Z","snapshot_observed_at":"2026-08-09T07:51:37.523530Z","submitted_at":"2025-08-06T13:03:21Z","title":"Thinking With Videos: Multimodal Tool-Augmented Reinforcement Learning for Long Video Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.04416","snapshot_observed_at":"2026-08-01T21:09:45.802892Z","title":"Thinking with videos: Multimodal tool-augmented reinforcement learning for long video reasoning.arXiv preprint arXiv:2508.04416, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:45.802892Z"},"links":{"cited_paper":"/paper/2508.04416","citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:6dd04f71dfde8da97995cd81379314b0d35d44bfd7f8d22019c77b1283882caa","observation_id":"1a0e3e4e-fe69-4804-b274-743ef21e3497","resolution":{"observed_at":"2026-08-01T21:09:45.802892Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.16852","last_updated":"2024-07-01T02:59:29Z","snapshot_observed_at":"2026-08-07T09:52:45.942315Z","submitted_at":"2024-06-24T17:58:06Z","title":"Long Context Transfer from Language to Vision","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.16852","snapshot_observed_at":"2026-08-01T21:09:45.855573Z","title":"Long context transfer from language to vision","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:45.855573Z"},"links":{"cited_paper":"/paper/2406.16852","citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:1d344f25cdb66bb54632f424470021abcead59e2f7c362e164f8f821520a91e1","observation_id":"d57e3b34-6e00-4cb7-8c0b-ecef8bc297ba","resolution":{"observed_at":"2026-08-01T21:09:45.855573Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04264","last_updated":"2025-01-01T15:53:58Z","snapshot_observed_at":"2026-08-03T20:38:36.602554Z","submitted_at":"2024-06-06T17:09:32Z","title":"MLVU: Benchmarking Multi-task Long Video Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04264","snapshot_observed_at":"2026-08-01T21:09:45.916527Z","title":"MLVU: A comprehensive benchmark for multi-task long video understanding.arXiv preprint arXiv:2406.04264, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:45.916527Z"},"links":{"cited_paper":"/paper/2406.04264","citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:00d60d06ec5937f4d33fc32c1d68264ec83c81ed2f015355bff2c5a3def40613","observation_id":"efea4c79-7408-4801-a2dd-e4c8152affec","resolution":{"observed_at":"2026-08-01T21:09:45.916527Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T21:09:45.965576Z","title":"User: Current Segment [t s-te]:","venue":null,"work_id":null,"year":2000},"citing_paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-01T21:09:45.965576Z"},"links":{"citing_paper":"/paper/2607.16189"},"observation_digest":"sha256:a875802a7f919e69615d3b4307050597b5d51949d57fae7a2704a2a29e6e522a","observation_id":"3752c4c4-f498-4b2e-8689-473e325b665f","resolution":{"observed_at":"2026-08-01T21:09:45.965576Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2607.16189","last_updated":"2026-07-17T17:59:27Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-08T05:22:33.805882Z","submitted_at":"2026-07-17T17:59:27Z","title":"Searching Videos as Trees: Self-Correcting Agents for Grounded Long Video QA"},"reference_resolution":{"displayed":44,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":43,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":44},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 44 of 44 outbound references and 0 inbound Pith citation observations for arXiv:2607.16189."}