{"as_of":"2026-08-18T15:04:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:f6256a1e04cdbfc1dbf1bfdbcf99e675907a3e6e4e4b71df243af4634b619a8e","coverage":[{"denominator":35,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":35,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T18:03:14.058499Z","state":"measured"},{"denominator":48,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":48,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-18T06:34:40.430872+00:00","state":"measured"},{"denominator":13,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":13,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T08:39:01.053662Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T01:19:20.288097Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.09313","snapshot_observed_at":"2026-08-04T08:39:01.053662Z","title":"Binfeng Xu, Zhiyuan PENG, Bowen Lei, Subhabrata Mukherjee, and Dongkuan Xu","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2510.19771","last_updated":"2026-07-07T17:34:06Z","snapshot_observed_at":"2026-08-15T20:31:50.112652Z","submitted_at":"2025-10-22T17:00:45Z","title":"Beyond Reactivity: Measuring Proactive Problem Solving in LLM Agents","version":4},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-04T08:39:01.053662Z"},"links":{"cited_paper":"/paper/2507.09313","citing_paper":"/paper/2510.19771"},"observation_digest":"sha256:6840e0b220eab12b435682f25487c4fc81678ad1296b270367a7b9ae9a3a88f1","observation_id":"5f6fd4f2-22b5-45a3-80e1-e540f88c8c65","resolution":{"observed_at":"2026-08-04T08:39:01.053662Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.09313","snapshot_observed_at":"2026-08-02T19:09:09.440908Z","title":"fill-with-silence","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.03447","last_updated":"2026-05-24T13:26:19Z","snapshot_observed_at":"2026-08-16T14:32:08.921682Z","submitted_at":"2026-03-03T19:02:46Z","title":"Proact-VL: A Proactive VideoLLM for Real-Time AI Companions","version":3},"reference_index":336,"source":"pdf_text","source_observed_at":"2026-08-02T19:09:09.440908Z"},"links":{"cited_paper":"/paper/2507.09313","citing_paper":"/paper/2603.03447"},"observation_digest":"sha256:43ba25840d2e5d48c4c0897d9407beb834c4915e27546d79591ad46e4523e76d","observation_id":"4027df28-ad4c-4b0f-917b-4e48460d6b75","resolution":{"observed_at":"2026-08-02T19:09:09.440908Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"cited_work":{"arxiv_id":"2507.09313","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.09313","snapshot_observed_at":"2026-07-04T01:19:20.288097Z","title":"Proactivevideoqa: A comprehensive benchmark evaluating proactive interactions in video large language models","venue":null,"work_id":"6c44dc17-6083-4f26-b7c6-e27f819dd061","year":2025},"citing_paper":{"arxiv_id":"2604.15037","last_updated":"2026-05-02T12:33:41Z","snapshot_observed_at":"2026-08-17T13:31:24.270554Z","submitted_at":"2026-04-16T14:06:30Z","title":"From Reactive to Proactive: Assessing the Proactivity of Voice Agents via ProVoice-Bench","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-10T11:44:12.373082Z"},"links":{"cited_paper":"/paper/2507.09313","citing_paper":"/paper/2604.15037"},"observation_digest":"sha256:56aee121ad47ba1118d92b60e6994a02335914c7ee64ffa6ead14b8b0048c308","observation_id":"4572f0d0-b18f-45ee-804b-a0cc2491ab37","resolution":{"observed_at":"2026-05-10T11:45:21.033091Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"cited_work":{"arxiv_id":"2507.09313","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.09313","snapshot_observed_at":"2026-07-04T01:19:20.288097Z","title":"Proactivevideoqa: A comprehensive benchmark evaluating proactive interactions in video large language models","venue":null,"work_id":"6c44dc17-6083-4f26-b7c6-e27f819dd061","year":2025},"citing_paper":{"arxiv_id":"2604.24317","last_updated":"2026-04-27T11:07:03Z","snapshot_observed_at":"2026-08-16T19:20:53.035767Z","submitted_at":"2026-04-27T11:07:03Z","title":"Don't Pause! Every prediction matters in a streaming video","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-05-08T04:32:01.379605Z"},"links":{"cited_paper":"/paper/2507.09313","citing_paper":"/paper/2604.24317"},"observation_digest":"sha256:bedfa3bd047ad9985f3e4fd72400da46cfbc8478d7986ac7ead80d637ab396ec","observation_id":"6c92c69d-1533-4d11-8cc6-ae801125e38f","resolution":{"observed_at":"2026-05-11T21:41:18.097956Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"cited_work":{"arxiv_id":"2507.09313","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.09313","snapshot_observed_at":"2026-07-04T01:19:20.288097Z","title":"Proactivevideoqa: A comprehensive benchmark evaluating proactive interactions in video large language models","venue":null,"work_id":"6c44dc17-6083-4f26-b7c6-e27f819dd061","year":2025},"citing_paper":{"arxiv_id":"2605.16381","last_updated":"2026-05-11T05:01:15Z","snapshot_observed_at":"2026-08-13T12:04:05.921038Z","submitted_at":"2026-05-11T05:01:15Z","title":"StreamPro: From Reactive Perception to Proactive Decision-Making in Streaming Video","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-20T23:32:38.332828Z"},"links":{"cited_paper":"/paper/2507.09313","citing_paper":"/paper/2605.16381"},"observation_digest":"sha256:bb503c39a5b7285d326259587c74f4f4cd6d7166f8961dc251516e67fe0355ff","observation_id":"068e4c34-be0c-4134-84b3-3b821d3b358c","resolution":{"observed_at":"2026-05-20T23:33:50.748812Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"cited_work":{"arxiv_id":"2507.09313","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.09313","snapshot_observed_at":"2026-07-04T01:19:20.288097Z","title":"Proactivevideoqa: A comprehensive benchmark evaluating proactive interactions in video large language models","venue":null,"work_id":"6c44dc17-6083-4f26-b7c6-e27f819dd061","year":2025},"citing_paper":{"arxiv_id":"2605.17360","last_updated":"2026-07-02T12:39:55Z","snapshot_observed_at":"2026-08-17T04:09:44.469533Z","submitted_at":"2026-05-17T09:57:01Z","title":"Omni-DuplexEval: Evaluating Real-time Duplex Omni-modal Interaction","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-20T13:36:44.071188Z"},"links":{"cited_paper":"/paper/2507.09313","citing_paper":"/paper/2605.17360"},"observation_digest":"sha256:f37feb11124ee991efe191d542401a03aa8de3f62689a13eb469431b13ff433f","observation_id":"41aead78-12f1-4da2-9363-84e7b519d662","resolution":{"observed_at":"2026-05-20T13:38:19.157208Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"cited_work":{"arxiv_id":"2507.09313","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.09313","snapshot_observed_at":"2026-07-04T01:19:20.288097Z","title":"Proactivevideoqa: A comprehensive benchmark evaluating proactive interactions in video large language models","venue":null,"work_id":"6c44dc17-6083-4f26-b7c6-e27f819dd061","year":2025},"citing_paper":{"arxiv_id":"2605.17360","last_updated":"2026-07-02T12:39:55Z","snapshot_observed_at":"2026-08-17T04:09:44.469533Z","submitted_at":"2026-05-17T09:57:01Z","title":"Omni-DuplexEval: Evaluating Real-time Duplex Omni-modal Interaction","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-07-04T01:11:42.073993Z"},"links":{"cited_paper":"/paper/2507.09313","citing_paper":"/paper/2605.17360"},"observation_digest":"sha256:41133a7258a3a900d1e3bd992e9a56a107ef407eeb1b9420269b9c1743b0f676","observation_id":"ca766167-9a2e-46cc-bc82-858fed26fffc","resolution":{"observed_at":"2026-07-04T01:19:20.291251Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"cited_work":{"arxiv_id":"2507.09313","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.09313","snapshot_observed_at":"2026-07-04T01:19:20.288097Z","title":"Proactivevideoqa: A comprehensive benchmark evaluating proactive interactions in video large language models","venue":null,"work_id":"6c44dc17-6083-4f26-b7c6-e27f819dd061","year":2025},"citing_paper":{"arxiv_id":"2605.17921","last_updated":"2026-06-01T06:29:58Z","snapshot_observed_at":"2026-08-15T01:27:26.647958Z","submitted_at":"2026-05-18T06:29:44Z","title":"An Efficient Streaming Video Understanding Framework with Agentic Control","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-05-20T11:30:22.151045Z"},"links":{"cited_paper":"/paper/2507.09313","citing_paper":"/paper/2605.17921"},"observation_digest":"sha256:c790afd85fcaf46a71f6ad1dca64d9dcb3c0cb65d2ee352219f434c5017196fb","observation_id":"fb22717f-9da6-4ae1-acb9-ffaba75f432d","resolution":{"observed_at":"2026-05-20T11:33:14.470074Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"cited_work":{"arxiv_id":"2507.09313","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.09313","snapshot_observed_at":"2026-07-04T01:19:20.288097Z","title":"Proactivevideoqa: A comprehensive benchmark evaluating proactive interactions in video large language models","venue":null,"work_id":"6c44dc17-6083-4f26-b7c6-e27f819dd061","year":2025},"citing_paper":{"arxiv_id":"2605.27074","last_updated":"2026-05-26T14:23:25Z","snapshot_observed_at":"2026-08-13T12:04:06.428850Z","submitted_at":"2026-05-26T14:23:25Z","title":"IPIBench: Evaluating Interactive Proactive Intelligence of MLLMs under Continuous Streams","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-06-29T18:24:57.881644Z"},"links":{"cited_paper":"/paper/2507.09313","citing_paper":"/paper/2605.27074"},"observation_digest":"sha256:115399894a67e7f73f48218082a04c1fc77bd80439265d825e77f49ecf0f267a","observation_id":"3e967234-cadc-4cee-95cf-5fbac879f8a2","resolution":{"observed_at":"2026-06-29T18:33:51.065824Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"cited_work":{"arxiv_id":"2507.09313","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.09313","snapshot_observed_at":"2026-07-04T01:19:20.288097Z","title":"Proactivevideoqa: A comprehensive benchmark evaluating proactive interactions in video large language models","venue":null,"work_id":"6c44dc17-6083-4f26-b7c6-e27f819dd061","year":2025},"citing_paper":{"arxiv_id":"2606.02482","last_updated":"2026-06-29T07:37:14Z","snapshot_observed_at":"2026-08-14T05:01:18.151983Z","submitted_at":"2026-06-01T16:52:11Z","title":"X-Stream: Exploring MLLMs as Multiplexers for Multi-Stream Understanding","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-06-28T15:06:22.102725Z"},"links":{"cited_paper":"/paper/2507.09313","citing_paper":"/paper/2606.02482"},"observation_digest":"sha256:0c1f82296a77b231395139ff36ec52c7ffbc7eeb2fa78a7661d011fdeb548e8a","observation_id":"c4af456a-a752-4d00-9ce7-3a177dc71ce5","resolution":{"observed_at":"2026-07-01T22:46:18.934183Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"cited_work":{"arxiv_id":"2507.09313","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.09313","snapshot_observed_at":"2026-07-04T01:19:20.288097Z","title":"Proactivevideoqa: A comprehensive benchmark evaluating proactive interactions in video large language models","venue":null,"work_id":"6c44dc17-6083-4f26-b7c6-e27f819dd061","year":2025},"citing_paper":{"arxiv_id":"2606.02482","last_updated":"2026-06-29T07:37:14Z","snapshot_observed_at":"2026-08-14T05:01:18.151983Z","submitted_at":"2026-06-01T16:52:11Z","title":"X-Stream: Exploring MLLMs as Multiplexers for Multi-Stream Understanding","version":3},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-06-30T10:42:37.401221Z"},"links":{"cited_paper":"/paper/2507.09313","citing_paper":"/paper/2606.02482"},"observation_digest":"sha256:db63e3023e4e08684bff33dab5f36a3d9c5325c35059334d318dc71a19835ccf","observation_id":"0dd45640-7b1f-485c-b538-17545bed0fe1","resolution":{"observed_at":"2026-06-30T10:44:36.467381Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.09313","snapshot_observed_at":"2026-07-12T05:36:40.278384Z","title":"arXiv preprint arXiv:2507.09313 (2025) 5, 8","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.02991","last_updated":"2026-07-03T06:00:04Z","snapshot_observed_at":"2026-08-06T10:48:23.317748Z","submitted_at":"2026-07-03T06:00:04Z","title":"GuideMe: Multi-Domain Task Guidance and Intervention in Streaming Video","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-07-12T05:36:40.278384Z"},"links":{"cited_paper":"/paper/2507.09313","citing_paper":"/paper/2607.02991"},"observation_digest":"sha256:c462e9ef22d9d99cd5c3520d4a1e4a0ef009af2db1a9a7e68443f44b5144f841","observation_id":"209f9f59-9164-42e5-8a37-7f88f5dc3007","resolution":{"observed_at":"2026-07-12T05:36:40.278384Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.09313","snapshot_observed_at":"2026-08-02T00:44:48.909159Z","title":"Proactivevideoqa: A comprehensive benchmark evaluating proactive interactions in video large language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14935","last_updated":"2026-07-16T12:47:59Z","snapshot_observed_at":"2026-08-12T23:43:17.706709Z","submitted_at":"2026-07-16T12:47:59Z","title":"VideoChat3: Fully Open Video MLLM for Efficient and Generalist Video Understanding","version":1},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-02T00:44:48.909159Z"},"links":{"cited_paper":"/paper/2507.09313","citing_paper":"/paper/2607.14935"},"observation_digest":"sha256:628a5de1aaf5c7abf4aeb55c60e001761a67cc81225a16463f45aa311c7e1330","observation_id":"7523c331-160b-4a39-b74a-911205ccb356","resolution":{"observed_at":"2026-08-02T00:44:48.909159Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2507.09313/citation-record","integrity":"/paper/2507.09313/integrity","json":"/paper/2507.09313/citation-record.json","paper":"/paper/2507.09313"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-08-14T04:17:22.593941Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.13923","snapshot_observed_at":"2026-08-06T18:03:13.969005Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:13.969005Z"},"links":{"cited_paper":"/paper/2502.13923","citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:05049254f2d18ebfad70e455e9fd0df61e6d6edb9a03bc4e6efb9c0caa5fbe2d","observation_id":"c6d6f3e2-9cf3-46ce-b728-a333e7d61e68","resolution":{"observed_at":"2026-08-06T18:03:13.969005Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.10818","last_updated":"2024-10-15T17:55:46Z","snapshot_observed_at":"2026-08-16T13:09:28.872428Z","submitted_at":"2024-10-14T17:59:58Z","title":"TemporalBench: Benchmarking Fine-grained Temporal Understanding for Multimodal Video Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.10818","snapshot_observed_at":"2026-08-06T18:03:13.972207Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:13.972207Z"},"links":{"cited_paper":"/paper/2410.10818","citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:0d157846f2e8aef424e66a5b02cb4e1e894a1f61ad7fe431b5af4c0852f5c978","observation_id":"e849daa4-bf69-475c-8e19-353675a6c28c","resolution":{"observed_at":"2026-08-06T18:03:13.972207Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:03:14.447012Z","title":null,"venue":null,"work_id":"17c4ed38-f30e-4934-80b1-87a9d7271af4","year":2024},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:13.975552Z"},"links":{"citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:7ac661dbbb7f62ce9ccea69493c4a70d0657e8e3322aaba7bf057a062a5f334f","observation_id":"10285eaf-f2e5-487f-b655-23d9189d1e48","resolution":{"observed_at":"2026-08-06T18:03:14.449302Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.05271","last_updated":"2025-09-26T12:52:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-06T18:57:08Z","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.05271","snapshot_observed_at":"2026-08-06T18:03:13.978291Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:13.978291Z"},"links":{"cited_paper":"/paper/2412.05271","citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:1c1a39fa036d25d839aec8e3570876b345934f5ba3dcf2c7c61f0191eed82608","observation_id":"486bec5f-f7ab-45ac-8921-1a765c1009d4","resolution":{"observed_at":"2026-08-06T18:03:13.978291Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:03:13.980956Z","title":null,"venue":null,"work_id":null,"year":1960},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:13.980956Z"},"links":{"citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:137b6ea6de8552b1295106c30ea98a114ecc487a6e82963a0b128326f56b7c44","observation_id":"9eac1249-9df0-4a71-9598-5a9417cdf906","resolution":{"observed_at":"2026-08-06T18:03:13.980956Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.14515","last_updated":"2024-10-30T13:38:10Z","snapshot_observed_at":"2026-08-17T20:22:23.176057Z","submitted_at":"2024-06-20T17:26:01Z","title":"MMBench-Video: A Long-Form Multi-Shot Benchmark for Holistic Video Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.14515","snapshot_observed_at":"2026-08-06T18:03:13.983556Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:13.983556Z"},"links":{"cited_paper":"/paper/2406.14515","citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:0f84d8f9777a327d13d02985c6975e540c682218994ca0e1c8b48af9a94609be","observation_id":"2b04f4a6-1cdd-49c6-a73d-34ebafa03d10","resolution":{"observed_at":"2026-08-06T18:03:13.983556Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.21075","last_updated":"2025-05-30T13:08:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-05-31T17:59:47Z","title":"Video-MME: The First-Ever Comprehensive Evaluation Benchmark of Multi-modal LLMs in Video Analysis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.21075","snapshot_observed_at":"2026-08-06T18:03:13.986518Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:13.986518Z"},"links":{"cited_paper":"/paper/2405.21075","citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:74d4a1c818d3dc30ec28ddf5d8c1660fd363d8707326d96df4e10f177081f786","observation_id":"dd5322e8-be16-40f6-9692-512e4e38a101","resolution":{"observed_at":"2026-08-06T18:03:13.986518Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:03:14.435352Z","title":null,"venue":null,"work_id":"00f34b96-ad5b-4e04-8f6c-a86b1c150932","year":2023},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:13.989094Z"},"links":{"citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:b23c66afed1cc241f44f32e0c8b87317943cadc53a8dfa7c03828ecf65b78830","observation_id":"1ba1b039-4af8-4801-a916-db5fd5cb4f99","resolution":{"observed_at":"2026-08-06T18:03:14.437624Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:03:14.428315Z","title":null,"venue":null,"work_id":"03141deb-6ae6-4728-8f0a-9a6fbd348d98","year":2023},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:13.991479Z"},"links":{"citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:7b04384e7c6f7aa840accca179944523c3f05e93f84340be7a3971f577a8bae0","observation_id":"5c973cca-26f8-45ae-802c-4afd7c5f51b5","resolution":{"observed_at":"2026-08-06T18:03:14.430735Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:03:14.421093Z","title":null,"venue":null,"work_id":"cf2f7378-3b50-4a4b-9f3e-c6a5178beac1","year":2018},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:13.993760Z"},"links":{"citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:21f4fe16ce014357ba4c225bd24a7c6065a1d4f9e2a8add5692d0fc582b42a5a","observation_id":"3a6bf101-c3d0-4f4c-9281-3e7231bac870","resolution":{"observed_at":"2026-08-06T18:03:14.423704Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03326","last_updated":"2024-10-26T16:35:13Z","snapshot_observed_at":"2026-08-18T11:56:50.710310Z","submitted_at":"2024-08-06T17:59:44Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03326","snapshot_observed_at":"2026-08-06T18:03:13.996017Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:13.996017Z"},"links":{"cited_paper":"/paper/2408.03326","citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:5cdb02c97eba2570a6a5333d73641601ec0ed10954a0c7b58274a1bf840af007","observation_id":"143a51d1-6530-49b9-8546-705b6ec35f4d","resolution":{"observed_at":"2026-08-06T18:03:13.996017Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16790","last_updated":"2024-04-25T17:39:35Z","snapshot_observed_at":"2026-08-16T13:57:47.881881Z","submitted_at":"2024-04-25T17:39:35Z","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16790","snapshot_observed_at":"2026-08-06T18:03:13.998784Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:13.998784Z"},"links":{"cited_paper":"/paper/2404.16790","citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:923a41ba8215967f01695116bfeaa84df4c21aed857aab4b52997371f615293f","observation_id":"0f0d8dc1-58fc-44ab-9453-12b949975ae9","resolution":{"observed_at":"2026-08-06T18:03:13.998784Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.16125","last_updated":"2023-08-02T08:02:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-30T04:25:16Z","title":"SEED-Bench: Benchmarking Multimodal LLMs with Generative Comprehension","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.16125","snapshot_observed_at":"2026-08-06T18:03:14.001294Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:14.001294Z"},"links":{"cited_paper":"/paper/2307.16125","citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:3ed5cc01437c90d1a767521c7dfe77f16bffda73d761db2020498ab541f670f0","observation_id":"52c34cc5-e012-4452-8163-c7f287325711","resolution":{"observed_at":"2026-08-06T18:03:14.001294Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.05579","last_updated":"2024-12-10T05:49:12Z","snapshot_observed_at":"2026-08-17T10:18:05.650518Z","submitted_at":"2024-12-07T08:07:24Z","title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.05579","snapshot_observed_at":"2026-08-06T18:03:14.003554Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:14.003554Z"},"links":{"cited_paper":"/paper/2412.05579","citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:fdded0613cde09efa1c3b3f954e15b063a4457c97542caacff3f9d4fa433cd04","observation_id":"00ce54f3-6064-41f4-8251-322351b7cbac","resolution":{"observed_at":"2026-08-06T18:03:14.003554Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:03:14.414008Z","title":null,"venue":null,"work_id":"77a3b0d4-72de-42d9-9f8e-3d10a8e89fbb","year":2023},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:14.005895Z"},"links":{"citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:f662ba26914afd2d9cd4847b1ef1d791ef294d6bfca502380e15e86b97fae27b","observation_id":"22fe2f4a-7f46-4fed-bf37-23f2ebc16f34","resolution":{"observed_at":"2026-08-06T18:03:14.416481Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:03:14.406346Z","title":null,"venue":null,"work_id":"77d7ef31-eaf7-4684-a839-9b628c57e881","year":2023},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:14.008278Z"},"links":{"citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:22d75b2166079e126cb9dd58dea9f683828481028be57aada249bde037e6e4c4","observation_id":"6822cf65-13fb-425d-801a-e6569e8fef1e","resolution":{"observed_at":"2026-08-06T18:03:14.408974Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.05510","last_updated":"2025-03-27T17:40:09Z","snapshot_observed_at":"2026-08-16T12:59:25.994981Z","submitted_at":"2025-01-09T19:00:01Z","title":"OVO-Bench: How Far is Your Video-LLMs from Real-World Online Video Understanding?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.05510","snapshot_observed_at":"2026-08-06T18:03:14.010398Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:14.010398Z"},"links":{"cited_paper":"/paper/2501.05510","citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:bca1de6bf4d3b548a98befd3ddbf381a6743a846adc2706b0774effbfc5bd4ee","observation_id":"467ce196-84d0-45fb-be40-1309c5607a5a","resolution":{"observed_at":"2026-08-06T18:03:14.010398Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:03:14.012841Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:14.012841Z"},"links":{"citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:2fd3cc4f4f49a3dfafe4a00091120379c57ced3f3da9c9bce0c1610e3752187a","observation_id":"ed36c7b1-e5cb-491c-b3c1-ecc56dfc9c8c","resolution":{"observed_at":"2026-08-06T18:03:14.012841Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-16T13:02:34.563787Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-06T18:03:14.015160Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:14.015160Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:1878d3b5a75b6f3aa55a81ff1eb30190fbf528ea6af2fb8c6997d30bea7c0cea","observation_id":"a49e3718-f95e-4026-a5f5-0ac0de340ec5","resolution":{"observed_at":"2026-08-06T18:03:14.015160Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.18111","last_updated":"2024-09-26T17:53:04Z","snapshot_observed_at":"2026-08-16T13:15:00.858953Z","submitted_at":"2024-09-26T17:53:04Z","title":"E.T. Bench: Towards Open-Ended Event-Level Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.18111","snapshot_observed_at":"2026-08-06T18:03:14.017689Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:14.017689Z"},"links":{"cited_paper":"/paper/2409.18111","citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:0f226a6e325fcb5119853df13604ed9e2a47a16c429253104a5f631437c094a9","observation_id":"420a5d6b-82a0-4e9c-97f4-e00fa50e180c","resolution":{"observed_at":"2026-08-06T18:03:14.017689Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:03:14.398809Z","title":null,"venue":null,"work_id":"5bed9404-017d-40dc-899d-b208fddd68b7","year":2002},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:14.020500Z"},"links":{"citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:3b4b047097b8eb9c41bf878bac580ed4798b7203bc0694c11ac88e6de90da177","observation_id":"ebc798c4-f520-4f9e-8b3f-6eef9486475d","resolution":{"observed_at":"2026-08-06T18:03:14.401107Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.03218","last_updated":"2025-01-06T18:55:10Z","snapshot_observed_at":"2026-08-16T12:59:35.733260Z","submitted_at":"2025-01-06T18:55:10Z","title":"Dispider: Enabling Video LLMs with Active Real-Time Interaction via Disentangled Perception, Decision, and Reaction","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.03218","snapshot_observed_at":"2026-08-06T18:03:14.023334Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:14.023334Z"},"links":{"cited_paper":"/paper/2501.03218","citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:8fef3eec72e46e5e94fe5da8f632acbdf5c8b94006e22063ae51546a97999156","observation_id":"7b0e7426-b2f9-414d-8ade-fed24f1340f5","resolution":{"observed_at":"2026-08-06T18:03:14.023334Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:03:14.391674Z","title":null,"venue":null,"work_id":"4c2581c0-ce77-41dd-b70e-d836476c40c7","year":2023},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:14.025661Z"},"links":{"citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:f46a8aa1e291c557eb6b3a8b176fa161500799017efa89561a0124a1e1576ea4","observation_id":"176d99bb-a4b2-4dbd-b5ed-83a21c18dae9","resolution":{"observed_at":"2026-08-06T18:03:14.393918Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:03:14.384382Z","title":null,"venue":null,"work_id":"17198c7e-9594-4966-b349-ec4a20ec7d98","year":2018},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:14.028068Z"},"links":{"citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:26e8f8965451a9ab302c6ecb4da2a57f31003570ad454366142af20f09b0ace1","observation_id":"f07fedcc-806e-4850-a9eb-4833e927b7d5","resolution":{"observed_at":"2026-08-06T18:03:14.386759Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:03:14.376352Z","title":"Lawrence Zitnick, and Devi Parikh","venue":null,"work_id":"d7a84671-142f-4fc8-8258-89bcfe7807fd","year":2014},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:14.030365Z"},"links":{"citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:c0271479d6566a8c92bad96d41f070a4994fdc1c41f30d479c7caf000c195d0d","observation_id":"5a549eee-007e-4704-830e-b56c9a39e8ce","resolution":{"observed_at":"2026-08-06T18:03:14.379613Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:03:14.032751Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:14.032751Z"},"links":{"citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:2bdd86ac82613bd7f1fa33cc33d57abdc0b9fdc52991d8f4df3ce866653dc0dc","observation_id":"998eb457-fc7c-4e94-9cfc-dd622733a284","resolution":{"observed_at":"2026-08-06T18:03:14.032751Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.22952","last_updated":"2025-03-29T02:46:58Z","snapshot_observed_at":"2026-08-16T12:45:48.372964Z","submitted_at":"2025-03-29T02:46:58Z","title":"OmniMMI: A Comprehensive Multi-modal Interaction Benchmark in Streaming Video Contexts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.22952","snapshot_observed_at":"2026-08-06T18:03:14.035067Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:14.035067Z"},"links":{"cited_paper":"/paper/2503.22952","citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:6dee96816d3a1357998bfbdcc1fe029a548d38d822c58f0d6b50a1b09c5f0e0c","observation_id":"8eb0e784-8b2e-4aff-8245-0d2dbdd6e43e","resolution":{"observed_at":"2026-08-06T18:03:14.035067Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.20215","last_updated":"2025-03-26T04:17:55Z","snapshot_observed_at":"2026-08-06T08:46:20.194739Z","submitted_at":"2025-03-26T04:17:55Z","title":"Qwen2.5-Omni Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.20215","snapshot_observed_at":"2026-08-06T18:03:14.037742Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:14.037742Z"},"links":{"cited_paper":"/paper/2503.20215","citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:61d8240c4d4a845cc542f20fb03adcc7fbc7dcf3d20e590e8badf350512e6518","observation_id":"85291699-b20d-458e-8586-54e298b76647","resolution":{"observed_at":"2026-08-06T18:03:14.037742Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.17343","last_updated":"2025-04-24T07:59:46Z","snapshot_observed_at":"2026-08-17T17:03:49.803696Z","submitted_at":"2025-04-24T07:59:46Z","title":"TimeChat-Online: 80% Visual Tokens are Naturally Redundant in Streaming Videos","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.17343","snapshot_observed_at":"2026-08-06T18:03:14.040372Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:14.040372Z"},"links":{"cited_paper":"/paper/2504.17343","citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:9809fa4daa94f0b9cde16b66c1f981f7d06c7ac22477bf842acc24ada5b13c40","observation_id":"46f8794c-4d6e-4420-8c39-7195e583ed0c","resolution":{"observed_at":"2026-08-06T18:03:14.040372Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:03:14.367696Z","title":null,"venue":null,"work_id":"11efbd8b-a044-4469-b487-2a160e20912f","year":2023},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:14.042972Z"},"links":{"citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:aa5bfc8a20e1ba9566db9d87d8c39a23ba6f9580e7e2081251c060ebc4834fcf","observation_id":"20623d6c-9e16-4f52-a6bc-4b75886e1e2c","resolution":{"observed_at":"2026-08-06T18:03:14.371311Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.03320","last_updated":"2024-07-03T17:59:21Z","snapshot_observed_at":"2026-08-14T16:53:05.474750Z","submitted_at":"2024-07-03T17:59:21Z","title":"InternLM-XComposer-2.5: A Versatile Large Vision Language Model Supporting Long-Contextual Input and Output","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.03320","snapshot_observed_at":"2026-08-06T18:03:14.045371Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:14.045371Z"},"links":{"cited_paper":"/paper/2407.03320","citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:dc5a1f17f55ab3b5f356ceabe5eb2ad430079be25618ec17e68142cb670324e7","observation_id":"209fa54e-22a5-428c-89cd-1c8afee2b76c","resolution":{"observed_at":"2026-08-06T18:03:14.045371Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.16852","last_updated":"2024-07-01T02:59:29Z","snapshot_observed_at":"2026-08-07T09:52:45.942315Z","submitted_at":"2024-06-24T17:58:06Z","title":"Long Context Transfer from Language to Vision","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.16852","snapshot_observed_at":"2026-08-06T18:03:14.047880Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:14.047880Z"},"links":{"cited_paper":"/paper/2406.16852","citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:d1afc3279c15376dbcb719d81d6f9919c1fd68a64276cb6aa037100ddd8f75a1","observation_id":"009ac0f9-9498-40c2-b9cd-b9a93ba48836","resolution":{"observed_at":"2026-08-06T18:03:14.047880Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1904.09675","last_updated":"2020-02-24T18:59:28Z","snapshot_observed_at":"2026-07-29T15:42:51.774083Z","submitted_at":"2019-04-21T23:08:53Z","title":"BERTScore: Evaluating Text Generation with BERT","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1904.09675","snapshot_observed_at":"2026-08-06T18:03:14.052687Z","title":"Weinberger, and Yoav Artzi","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:14.052687Z"},"links":{"cited_paper":"/paper/1904.09675","citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:7448475d1ca6270c6f2b8d72a45fd1e17ec526db9b3d459840f1a94d9863a5ea","observation_id":"7a6aeb78-0594-46dc-9dd5-0891c5923a75","resolution":{"observed_at":"2026-08-06T18:03:14.052687Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:03:14.055616Z","title":"online\" 'onlinestring :=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:14.055616Z"},"links":{"citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:e05d46236794ded25d31380d8de1c2d4b759f04841fe8e69e6d19eb6a5c7967b","observation_id":"8fd3de2e-9751-44be-9eab-992a8c16393c","resolution":{"observed_at":"2026-08-06T18:03:14.055616Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:03:14.058499Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models","version":2},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-06T18:03:14.058499Z"},"links":{"citing_paper":"/paper/2507.09313"},"observation_digest":"sha256:12b032e9778a8fb5056277d455888b945f97b91fa8d627b14cd1671fe4b6045b","observation_id":"6e4d065e-de94-4dc6-b3b2-278454a4024a","resolution":{"observed_at":"2026-08-06T18:03:14.058499Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2507.09313","last_updated":"2025-07-15T11:48:07Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-17T10:19:04.265079Z","submitted_at":"2025-07-12T15:11:50Z","title":"ProactiveVideoQA: A Comprehensive Benchmark Evaluating Proactive Interactions in Video Large Language Models"},"reference_resolution":{"displayed":35,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":34,"verified_exact":0,"verified_fuzzy":1},"total_outbound_references":35},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 35 of 35 outbound references and 13 inbound Pith citation observations for arXiv:2507.09313."}