{"as_of":"2026-08-08T11:19:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:33ad28a9973edbd1719c6d05e4c6630c5ec158c78f338e58dc266a8596b9d5f6","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":50,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":50,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":50,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":50,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:42:56.680449Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T20:00:08.182505Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2408.10188","last_updated":"2024-12-13T02:32:06Z","snapshot_observed_at":"2026-08-05T14:57:53.592979Z","submitted_at":"2024-08-19T17:48:08Z","title":"LongVILA: Scaling Long-Context Visual Language Models for Long Videos","version":6},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-17T03:51:25.396887Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2408.10188"},"observation_digest":"sha256:2f307adaa0424216ed0ac4dea9267c4b0733dda4409b518cc9e56a6af00fde96","observation_id":"16ce91fa-49dc-4fb9-85aa-a081ec43c810","resolution":{"observed_at":"2026-05-17T03:51:25.463933Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2411.02327","last_updated":"2026-05-01T17:44:24Z","snapshot_observed_at":"2026-07-06T19:45:00.034364Z","submitted_at":"2024-11-04T17:50:36Z","title":"PPLLaVA: Varied Video Sequence Understanding With Prompt Guidance","version":4},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-23T17:31:59.030963Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2411.02327"},"observation_digest":"sha256:b58047323ee9191581487e5761d1f558265e0f7f664032981dda1fe875570a7f","observation_id":"fed152d1-acea-4a91-8536-678bbe328e50","resolution":{"observed_at":"2026-05-23T17:33:15.672665Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2501.13106","last_updated":"2025-06-03T03:33:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T18:59:46Z","title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","version":4},"reference_index":168,"source":"pdf_text","source_observed_at":"2026-05-11T01:19:59.603343Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2501.13106"},"observation_digest":"sha256:c8c1a6444cac0f0bc48ac8a4605dc87e4f92326fba7383c8b3ccb490fe6bc892","observation_id":"80098db2-8922-4721-9090-bfaa8ada79d8","resolution":{"observed_at":"2026-05-11T01:20:00.236233Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-08-07T15:42:56.680449Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14231","last_updated":"2025-05-20T11:40:43Z","snapshot_observed_at":"2026-08-07T15:35:48.142895Z","submitted_at":"2025-05-20T11:40:43Z","title":"UniVG-R1: Reasoning Guided Universal Visual Grounding with Reinforcement Learning","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-07T15:42:56.680449Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2505.14231"},"observation_digest":"sha256:98730d1f9973c74497b9610d33c308bdf9c0d4097b2e4fcca29094e354f45eae","observation_id":"739c46fe-1d13-4e13-b3db-19f83f83e070","resolution":{"observed_at":"2026-08-07T15:42:56.680449Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2505.15269","last_updated":"2026-04-23T12:54:38Z","snapshot_observed_at":"2026-08-03T01:07:06.468931Z","submitted_at":"2025-05-21T08:47:15Z","title":"LiveVLM: Efficient Online Video Understanding via Streaming-Oriented KV Cache and Retrieval","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-22T14:26:59.015559Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2505.15269"},"observation_digest":"sha256:08eabf6053a138e02933638ba7c70657c470775f830c95a52668cb64d769f5f2","observation_id":"9c3d523b-1f4c-4892-95fa-71444da65956","resolution":{"observed_at":"2026-05-22T14:31:40.813023Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2505.18719","last_updated":"2025-05-24T14:42:51Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-24T14:42:51Z","title":"VLA-RL: Towards Masterful and General Robotic Manipulation with Scalable Reinforcement Learning","version":1},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-05-16T12:55:40.245908Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2505.18719"},"observation_digest":"sha256:51247d023a8e1427f4c7cc256c51f2013e6d0bd2b313df027b6d3ad06260c357","observation_id":"f614c2f9-c1ee-4bd9-92c9-241b1d2d38bb","resolution":{"observed_at":"2026-05-16T12:55:40.390683Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-08-07T12:40:46.869024Z","title":"Zhang, Y","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23922","last_updated":"2025-05-29T18:15:07Z","snapshot_observed_at":"2026-08-07T12:35:51.712975Z","submitted_at":"2025-05-29T18:15:07Z","title":"ScaleLong: A Multi-Timescale Benchmark for Long Video Understanding","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-07T12:40:46.869024Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2505.23922"},"observation_digest":"sha256:5368588f035974b1f6e9d8c41ff4878b1231a67846b4f5cd6c17ebe9b0218648","observation_id":"c6df2d43-09d9-4618-97e4-0c342718411e","resolution":{"observed_at":"2026-08-07T12:40:46.869024Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-08-07T05:49:53.419480Z","title":"Flash- vstream: Memory-based real-time understanding for long video streams.arXiv preprint arXiv:2406.08085, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.07016","last_updated":"2025-06-13T19:05:47Z","snapshot_observed_at":"2026-08-07T11:48:30.543139Z","submitted_at":"2025-06-08T06:34:29Z","title":"MAGNET: A Multi-agent Framework for Finding Audio-Visual Needles by Reasoning over Multi-Video Haystacks","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-07T05:49:53.419480Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2506.07016"},"observation_digest":"sha256:cd9250ada6b6c604c76d9d69d5242e44f05e6bb30696c950ec2bd72b0756d772","observation_id":"5c7ac51c-e03f-4c20-be75-737f1f5e4b4e","resolution":{"observed_at":"2026-08-07T05:49:53.419480Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-08-06T23:50:23.560907Z","title":"Flash-vstream: Memory- based real-time understanding for long video streams","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.16058","last_updated":"2025-06-24T03:11:42Z","snapshot_observed_at":"2026-08-06T23:41:59.563520Z","submitted_at":"2025-06-19T06:32:53Z","title":"Stepping Out of Similar Semantic Space for Open-Vocabulary Segmentation","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-06T23:50:23.560907Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2506.16058"},"observation_digest":"sha256:990990e2a0bb18ca069e75b1eccb92e192812f37ac4299ac31e16f7ceb5c1bc1","observation_id":"fe2be44c-1e16-44a6-ad2d-f8e543fb0667","resolution":{"observed_at":"2026-08-06T23:50:23.560907Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-08-06T18:06:18.952812Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09217","last_updated":"2025-07-12T09:24:28Z","snapshot_observed_at":"2026-08-07T15:17:00.421360Z","submitted_at":"2025-07-12T09:24:28Z","title":"Online Long-term Point Tracking in the Foundation Model Era","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T18:06:18.952812Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2507.09217"},"observation_digest":"sha256:3d1a304f336d170b1eed9993f45635dbbd405f40a030d6822a58a010cc9b1d1b","observation_id":"8e73ad35-a823-4d03-a8a7-bb0f21cdc78c","resolution":{"observed_at":"2026-08-06T18:06:18.952812Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-08-06T00:00:03.777349Z","title":"Flash-vstream: Memory- based real-time understanding for long video streams.arXiv preprint arXiv:2406.08085, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04546","last_updated":"2025-08-06T15:33:49Z","snapshot_observed_at":"2026-08-07T05:03:50.492168Z","submitted_at":"2025-08-06T15:33:49Z","title":"Hierarchical Event Memory for Accurate and Low-latency Online Video Temporal Grounding","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T00:00:03.777349Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2508.04546"},"observation_digest":"sha256:92f7ac4785ce0e4b468bd8112333115d4d90970300cdcdec04846d04c477674e","observation_id":"2ba91c3e-102d-4de4-a5c1-c99f9b4b88de","resolution":{"observed_at":"2026-08-06T00:00:03.777349Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-08-03T20:15:36.429995Z","title":"Flash-vstream: Memory- based real-time understanding for long video streams.arXiv preprint arXiv:2406.08085, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2511.20644","last_updated":"2026-07-09T17:59:10Z","snapshot_observed_at":"2026-08-06T01:52:36.714650Z","submitted_at":"2025-11-25T18:59:02Z","title":"Vision-Language Memory for Spatial Reasoning","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-03T20:15:36.429995Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2511.20644"},"observation_digest":"sha256:087a38869bdb284e5896ac1da2f6a112e65205c15442b0bb12ec2742cee6a1a9","observation_id":"946eac70-1f6c-4433-aad0-c82da8c1daea","resolution":{"observed_at":"2026-08-03T20:15:36.429995Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2511.21998","last_updated":"2026-04-12T05:29:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-11-27T00:54:35Z","title":"Can Multi-Modal LLMs Provide Live Step-by-Step Task Guidance?","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-17T05:36:09.208754Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2511.21998"},"observation_digest":"sha256:bc8adb7f63c16b7ffb3107a8a5e092a13bb0eefee893fb897293174ab858a225","observation_id":"14008280-962a-4104-a8cc-86e1bb31ca92","resolution":{"observed_at":"2026-05-17T05:39:06.167340Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2512.21334","last_updated":"2026-04-10T15:00:46Z","snapshot_observed_at":"2026-08-03T08:10:30.527963Z","submitted_at":"2025-12-24T18:59:36Z","title":"Streaming Video Instruction Tuning","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-16T19:44:11.032898Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2512.21334"},"observation_digest":"sha256:ac0c3884b0d28c4ec3e7c3e50a890dde2b682a76d6940cacf70f641b374a1766","observation_id":"394e8922-e627-4b86-b997-907e05b8b432","resolution":{"observed_at":"2026-05-16T19:48:21.792961Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2601.14724","last_updated":"2026-05-07T12:10:26Z","snapshot_observed_at":"2026-08-05T03:02:47.238051Z","submitted_at":"2026-01-21T07:26:15Z","title":"HERMES: KV Cache as Hierarchical Memory for Efficient Streaming Video Understanding","version":4},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-16T12:55:04.564442Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2601.14724"},"observation_digest":"sha256:3ab68f307d66dd67cd10e855607c527ef989b71e9d9468f6ab3290a19f6dde8d","observation_id":"22a96c46-ffa6-459c-a1f9-b95b727ad2cb","resolution":{"observed_at":"2026-05-16T12:57:53.864829Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2602.20913","last_updated":"2026-04-15T16:09:22Z","snapshot_observed_at":"2026-08-02T12:38:41.181077Z","submitted_at":"2026-02-24T13:49:47Z","title":"LongVideo-R1: Smart Navigation for Low-cost Long Video Understanding","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-15T20:01:31.129959Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2602.20913"},"observation_digest":"sha256:410bc0b9da2117702be158392c041a5eb29c31582a2ee786a4a4aea94a2b585a","observation_id":"0bf7a826-66ac-4eb0-a31b-6e0dc62ccb89","resolution":{"observed_at":"2026-05-15T20:01:33.505455Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2603.01455","last_updated":"2026-04-21T05:06:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-03-02T05:12:45Z","title":"From Verbatim to Gist: Distilling Pyramidal Multimodal Memory via Semantic Information Bottleneck for Long-Horizon Video Agents","version":3},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-15T18:09:59.236030Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2603.01455"},"observation_digest":"sha256:1597f054418197b6f2bd5052d24e058d9c78ddba834911243df5379db661e28b","observation_id":"5bc8b24f-ac61-4f4b-a48e-e571ca4389d6","resolution":{"observed_at":"2026-05-15T18:10:13.107496Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-14T22:25:30.596294Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.12219","last_updated":"2026-06-22T03:37:19Z","snapshot_observed_at":"2026-07-14T22:25:30.331088Z","submitted_at":"2026-03-12T17:44:27Z","title":"An Updated SynthPop Model for Microlensing Simulations I: Model Description & Evaluation","version":3},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-07-14T22:25:30.596294Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2603.12219"},"observation_digest":"sha256:51974fb0d53367a42c65d21d9b13afdb01151fb4245bdbc31dcf8754eafd273d","observation_id":"104221c3-5d21-4f4e-a341-9336cef722da","resolution":{"observed_at":"2026-07-14T22:25:30.596294Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2603.27259","last_updated":"2026-06-18T21:01:40Z","snapshot_observed_at":"2026-08-02T11:52:58.572026Z","submitted_at":"2026-03-28T12:44:19Z","title":"Seeing the Scene Matters: Revealing Forgetting in Video Understanding Models with a Scene-Aware Long-Video Benchmark","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-14T22:05:07.326202Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2603.27259"},"observation_digest":"sha256:623aef466ad6242d80adae9177689d1678b39e0c57e1628748d8ca79fcf57acd","observation_id":"df9ac8cf-62b0-49a1-b05c-90746584bb94","resolution":{"observed_at":"2026-05-14T22:08:04.448097Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2604.07634","last_updated":"2026-05-05T18:37:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-08T22:31:20Z","title":"VSAS-Bench: Real-Time Evaluation of Visual Streaming Assistant Models","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-10T17:39:30.731688Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2604.07634"},"observation_digest":"sha256:0c951c0116ccfdb1d4d03bb420ab66ba3c4ee9b1c2aa18065ff707a1121f5938","observation_id":"2583ddc4-2dc7-4abc-af36-05ebbeaae521","resolution":{"observed_at":"2026-05-11T06:25:58.031174Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2604.09000","last_updated":"2026-04-23T08:05:34Z","snapshot_observed_at":"2026-07-06T22:57:59.555972Z","submitted_at":"2026-04-10T06:11:34Z","title":"StreamMeCo: Long-Term Agent Memory Compression for Efficient Streaming Video Understanding","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-10T18:19:41.543165Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2604.09000"},"observation_digest":"sha256:c67a41021e248c33c33040b02e960e9481dbb9d20df961bc7eea74d24de8211e","observation_id":"8cb884b5-c0e4-4c76-b400-7aa8cb670cf2","resolution":{"observed_at":"2026-05-11T00:45:50.436845Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2604.17052","last_updated":"2026-04-18T16:22:05Z","snapshot_observed_at":"2026-08-02T06:02:58.067734Z","submitted_at":"2026-04-18T16:22:05Z","title":"OASIS: On-Demand Hierarchical Event Memory for Streaming Video Reasoning","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-10T06:51:52.861981Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2604.17052"},"observation_digest":"sha256:b40d81c0259dfd100d9569e2dd0fd38739f7b04e4ab772e834cd8cfa8cb2f08c","observation_id":"266777e1-7e61-40fb-aea8-674cb00bbeca","resolution":{"observed_at":"2026-05-10T06:56:47.788325Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2604.24317","last_updated":"2026-04-27T11:07:03Z","snapshot_observed_at":"2026-07-06T23:10:24.992210Z","submitted_at":"2026-04-27T11:07:03Z","title":"Don't Pause! Every prediction matters in a streaming video","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-05-08T04:32:01.379605Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2604.24317"},"observation_digest":"sha256:f028f61ed8e067b548a1af7b5dcfd138ef2740497cb179a9149d8893d71d57b6","observation_id":"13efc00b-3054-4d1d-8fd7-92eba1529fe7","resolution":{"observed_at":"2026-05-11T21:41:18.019791Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-07-06T23:15:01.968194Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:43b48d69d1e811ea66930600f042a50a8cdab3d606a7718c7b00b7af6cd2530d","observation_id":"09c9c0bb-1fff-461f-aa8e-83141dc06e75","resolution":{"observed_at":"2026-05-11T11:31:03.562532Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2605.04515","last_updated":"2026-05-06T05:48:56Z","snapshot_observed_at":"2026-08-06T20:21:26.793848Z","submitted_at":"2026-05-06T05:48:56Z","title":"From Priors to Perception: Grounding Video-LLMs in Physical Reality","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-08T17:41:23.233366Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2605.04515"},"observation_digest":"sha256:31fc310c6e12878333300bc63f1b0f77e59469f5621c56ff2a63997bb6206f71","observation_id":"5cdcc589-c2b1-4b8e-8faa-0b46fd2d4312","resolution":{"observed_at":"2026-05-11T17:21:08.334591Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2605.15054","last_updated":"2026-05-14T16:48:03Z","snapshot_observed_at":"2026-07-06T23:26:23.179482Z","submitted_at":"2026-05-14T16:48:03Z","title":"LATERN: Test-Time Context-Aware Explainable Video Anomaly Detection","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-06-30T21:19:12.655706Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2605.15054"},"observation_digest":"sha256:3c2afbbb8563bc1b83590691562e8a585d5f3443bcfd864bcea7b9712a5f3217","observation_id":"212f5093-1618-4c58-a2ad-0068af94c72a","resolution":{"observed_at":"2026-07-01T14:25:47.016973Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2605.17065","last_updated":"2026-05-16T16:15:59Z","snapshot_observed_at":"2026-08-02T03:22:29.983512Z","submitted_at":"2026-05-16T16:15:59Z","title":"PyraVid: Hierarchical Multimodal Memory for Long-Horizon Video Reasoning","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-05-20T15:12:00.408851Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2605.17065"},"observation_digest":"sha256:db43c2b792e51bfa8587c1645f5a46fedcca63c6f2f945c8b9a6ae5a1d7dec3e","observation_id":"eb91be19-68ab-41e9-ba79-cff4ed9f0561","resolution":{"observed_at":"2026-05-20T15:13:24.798248Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2605.17283","last_updated":"2026-05-17T06:39:05Z","snapshot_observed_at":"2026-08-02T16:55:37.869194Z","submitted_at":"2026-05-17T06:39:05Z","title":"OProver: A Unified Framework for Agentic Formal Theorem Proving","version":1},"reference_index":80,"source":"arxiv_source","source_observed_at":"2026-05-20T14:43:46.517807Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2605.17283"},"observation_digest":"sha256:3ce27e6b38a41931704f7f721d5b0ee8d81c1bceca773308eda4d12de5059054","observation_id":"0c263db3-57e5-4683-92be-a303128f8bbc","resolution":{"observed_at":"2026-05-20T14:48:23.506925Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2605.17360","last_updated":"2026-07-02T12:39:55Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T09:57:01Z","title":"Omni-DuplexEval: Evaluating Real-time Duplex Omni-modal Interaction","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-20T13:36:44.071188Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2605.17360"},"observation_digest":"sha256:6895cae7e8a77bc87733e72957cbe23d5e847428503071b5da2a4a62741ddd6c","observation_id":"96f99809-c799-4cf6-8451-346d8e4bcfc1","resolution":{"observed_at":"2026-05-20T13:38:19.169529Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2605.17360","last_updated":"2026-07-02T12:39:55Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T09:57:01Z","title":"Omni-DuplexEval: Evaluating Real-time Duplex Omni-modal Interaction","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-07-04T01:11:42.073993Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2605.17360"},"observation_digest":"sha256:ed356371510490d6c05b0896b8acbb5e2356fc80f7de4ac16f45d6a7737cee4f","observation_id":"aaf4586a-8910-4338-80b9-98681305850c","resolution":{"observed_at":"2026-07-04T01:19:20.326643Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2605.27074","last_updated":"2026-05-26T14:23:25Z","snapshot_observed_at":"2026-08-02T20:44:04.329111Z","submitted_at":"2026-05-26T14:23:25Z","title":"IPIBench: Evaluating Interactive Proactive Intelligence of MLLMs under Continuous Streams","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-06-29T18:24:57.881644Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2605.27074"},"observation_digest":"sha256:1901917c5e784cff8218c87cdf19407fa70c1236c67166e28e9163e64a574b1c","observation_id":"6743be0a-b7b7-40fd-9707-80343302ea06","resolution":{"observed_at":"2026-06-29T18:33:51.028063Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2605.31598","last_updated":"2026-05-29T17:59:02Z","snapshot_observed_at":"2026-08-07T05:09:45.280725Z","submitted_at":"2026-05-29T17:59:02Z","title":"Linear Scaling Video VLMs for Long Video Understanding","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-06-28T23:00:11.246232Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2605.31598"},"observation_digest":"sha256:1fa11f48663fbf5d8f56bdf5e24fda6b38007057d3a2e487adfbbfa911f1f681","observation_id":"2b18fe52-a846-4236-b189-fc792d0a4fb2","resolution":{"observed_at":"2026-06-28T23:02:46.258016Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2606.03100","last_updated":"2026-06-04T05:13:15Z","snapshot_observed_at":"2026-08-01T16:37:35.015535Z","submitted_at":"2026-06-02T03:38:51Z","title":"Zero-Shot 3D Question Answering via Hierarchical View-to-Token Transportation","version":2},"reference_index":81,"source":"arxiv_source","source_observed_at":"2026-06-28T10:54:02.188634Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2606.03100"},"observation_digest":"sha256:e6e8a5eedfd54704392f66000d447e559583e4b183fc37429de884b9b02b20d6","observation_id":"7356d735-315d-4c30-af27-76028d7db25c","resolution":{"observed_at":"2026-07-02T02:26:27.065722Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2606.03890","last_updated":"2026-06-02T16:51:32Z","snapshot_observed_at":"2026-07-06T23:44:05.167767Z","submitted_at":"2026-06-02T16:51:32Z","title":"OVO-S-Bench: A Hierarchical Benchmark for Streaming Spatial Intelligence in Multimodal LLMs","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-06-28T11:02:07.122615Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2606.03890"},"observation_digest":"sha256:da81ac9f3e19dc8ec070237189f5d7b98f35a40248413d62947291267e52af4b","observation_id":"cb654ab0-e384-4fd2-9d22-52721b12820f","resolution":{"observed_at":"2026-07-02T02:16:27.100649Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:cf6be776a9e42980ce972489dc34799eaafaf06f8b89cfeb8636b42506bc638b","observation_id":"f17c9624-dcc6-4b32-9028-2268977d50fc","resolution":{"observed_at":"2026-07-02T17:07:12.826114Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2606.07433","last_updated":"2026-06-05T16:29:13Z","snapshot_observed_at":"2026-08-01T21:05:06.439607Z","submitted_at":"2026-06-05T16:29:13Z","title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-27T22:00:28.350003Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2606.07433"},"observation_digest":"sha256:c31494d21f8a1cf69a203606d91b8f9e8dba16bdeb018484d89829ba26c9534e","observation_id":"ace06fdb-019e-44c5-aecf-8cefd999604a","resolution":{"observed_at":"2026-07-02T17:27:15.497535Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2606.09547","last_updated":"2026-06-17T20:05:01Z","snapshot_observed_at":"2026-08-07T22:14:54.132288Z","submitted_at":"2026-06-08T14:27:20Z","title":"Streaming Interventions: Can Video Large Language Models Correct Mistakes as They Occur?","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-06-27T16:52:22.811857Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2606.09547"},"observation_digest":"sha256:1adeb18ac1bc10abaabe48864b3a18d25c8f3e5c3bfbcca20e28bdf16032ea70","observation_id":"8b545684-c27d-47da-8667-b0cb0d11a8eb","resolution":{"observed_at":"2026-07-03T00:57:30.647830Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-08-08T11:16:26.006355Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:7381b344e40c6ded1ff499ce7526dda1751464fc9a4458177f48b2c18c76b83d","observation_id":"8b069b59-ddbf-4ffd-8919-b3520863a203","resolution":{"observed_at":"2026-07-03T20:38:56.143340Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2606.19849","last_updated":"2026-06-18T06:57:31Z","snapshot_observed_at":"2026-08-06T15:40:32.013079Z","submitted_at":"2026-06-18T06:57:31Z","title":"ViCoStream: Streaming VideoLLMs Can Run Beyond 100 FPS with Stage-Wise Coordinated Inference","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-06-26T18:17:53.013043Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2606.19849"},"observation_digest":"sha256:39b7f1c3aff8429d4968a14c6e8a8ebc81075ce9dab505e7de14d6ef769c23ea","observation_id":"587d4ed6-2de8-4cf1-aff5-54b31c752bf1","resolution":{"observed_at":"2026-07-04T03:19:29.905981Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2606.20726","last_updated":"2026-06-17T03:30:01Z","snapshot_observed_at":"2026-08-06T05:29:51.934551Z","submitted_at":"2026-06-17T03:30:01Z","title":"How Well Can Your Video Model Remember? Measuring Memory-Budget Trade-offs in Long Video Understanding","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-06-26T21:51:04.050833Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2606.20726"},"observation_digest":"sha256:751506c18854534b1ec4a074882e22dee72515d215569e185c7523582f130c33","observation_id":"25d70545-f7d4-4401-ad83-bd7f31670d73","resolution":{"observed_at":"2026-07-03T23:39:04.710461Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2606.24477","last_updated":"2026-06-23T12:13:19Z","snapshot_observed_at":"2026-07-06T23:59:03.724013Z","submitted_at":"2026-06-23T12:13:19Z","title":"video-SALMONN-R$^3$: Learning to ReWatch, ReAsk, and ReAnswer for Efficient Video Understanding","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-06-26T00:19:26.153682Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2606.24477"},"observation_digest":"sha256:b8367e4d65ff03468f40931f6a0f79436cc9e5706ce0dab2f3137703570ec69e","observation_id":"16e9037b-f4fe-44d4-8fa8-a2fee29d1c6c","resolution":{"observed_at":"2026-07-04T16:39:58.343073Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2606.25658","last_updated":"2026-06-24T10:11:08Z","snapshot_observed_at":"2026-08-07T02:41:08.601257Z","submitted_at":"2026-06-24T10:11:08Z","title":"Towards a Dynamic and Fixed-budget Memory Bank for Efficient Streaming Video Understanding","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-06-25T20:54:49.319252Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2606.25658"},"observation_digest":"sha256:77d5e9d25e6e4b406dd0ab6987ba144018a5636356bc73f2a125bb3bb5338fe4","observation_id":"bf1fbb9e-037d-40dd-9ba6-a6f5309f6bd5","resolution":{"observed_at":"2026-07-04T20:00:08.184260Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-11T06:35:35.951554Z","title":"Flash-vstream: Memory- based real-time understanding for long video streams.arXiv preprint arXiv:2406.08085, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.05511","last_updated":"2026-07-06T18:00:06Z","snapshot_observed_at":"2026-08-06T10:36:18.804889Z","submitted_at":"2026-07-06T18:00:06Z","title":"Light-Omni: Reflex over Reasoning in Agentic Video Understanding with Long-Term Memory","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-07-11T06:35:35.951554Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2607.05511"},"observation_digest":"sha256:39dcc7620d2ca5dabf8e9066565dcda14c74d8d195383bd711e04a1a143bb3a6","observation_id":"97a1906b-585e-4bfb-9a68-31c654f7fd22","resolution":{"observed_at":"2026-07-11T06:35:35.951554Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-14T05:01:06.200663Z","title":"arXiv preprint arXiv:2406.08085 (2024) 2, 4","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.11523","last_updated":"2026-07-13T13:09:45Z","snapshot_observed_at":"2026-08-06T02:25:09.480384Z","submitted_at":"2026-07-13T13:09:45Z","title":"Vinci2: Providing Proactive Assistance in Continuous Egocentric Videos","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-07-14T05:01:06.200663Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2607.11523"},"observation_digest":"sha256:1075049bc20bf96bb4027f28386c824ba35e26bf6ed39bd24b405eb25fe1461c","observation_id":"8226f807-a8ea-4980-a197-0d07ec3918a5","resolution":{"observed_at":"2026-07-14T05:01:06.200663Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-08-02T00:44:41.589936Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.14935","last_updated":"2026-07-16T12:47:59Z","snapshot_observed_at":"2026-08-06T19:13:04.526298Z","submitted_at":"2026-07-16T12:47:59Z","title":"VideoChat3: Fully Open Video MLLM for Efficient and Generalist Video Understanding","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-02T00:44:41.589936Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2607.14935"},"observation_digest":"sha256:6af7c9857416d6103896939be4da1d541a32fa01c9a789c39545411e73d35740","observation_id":"7e5440df-9915-40ca-8717-78f8973438bf","resolution":{"observed_at":"2026-08-02T00:44:41.589936Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-31T06:20:13.884368Z","title":"Flash-VStream: Memory-based real-time understanding for long video streams.arXiv preprint arXiv:2406.08085, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24904","last_updated":"2026-07-27T17:59:53Z","snapshot_observed_at":"2026-08-07T06:47:24.083796Z","submitted_at":"2026-07-27T17:59:53Z","title":"Mage-VL: An Efficient Codec-Native Streaming Multimodal Foundation Model","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-07-31T06:20:13.884368Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2607.24904"},"observation_digest":"sha256:3082bd718b23657196bd323ad7da62ff467ef2e27eeae1b61685d24ca281b556","observation_id":"d760ee98-5e90-420d-9bca-e53aecc7cd85","resolution":{"observed_at":"2026-07-31T06:20:13.884368Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-08-04T03:21:48.375332Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.28312","last_updated":"2026-08-01T16:00:11Z","snapshot_observed_at":"2026-08-06T23:11:29.524233Z","submitted_at":"2026-07-30T14:47:00Z","title":"ObjectStream: Latent Objects as Memory Anchors for Streaming Video Understanding","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-04T03:21:48.375332Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2607.28312"},"observation_digest":"sha256:4ee6cb03b7bb3eed9f60b3d77991f06fe30925b061c5a65e00d8dde54701389e","observation_id":"1b043da0-8aaa-49d4-99f7-8bb3ff0b33e3","resolution":{"observed_at":"2026-08-04T03:21:48.375332Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-08-03T00:45:21.931265Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.28678","last_updated":"2026-07-29T10:25:23Z","snapshot_observed_at":"2026-08-07T03:57:17.672794Z","submitted_at":"2026-07-29T10:25:23Z","title":"ViSAGE: Constructing Self-Correcting Memories for Long-Form Video Understanding","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-03T00:45:21.931265Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2607.28678"},"observation_digest":"sha256:a8e5797155458344cdb136679da4a3a185c90f84520d7af9af6c62d1ea786309","observation_id":"9cea1237-dc36-4c46-8a46-6741e1a482ff","resolution":{"observed_at":"2026-08-03T00:45:21.931265Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-08-04T08:12:47.458517Z","title":"Zhang, X.; Jia, Z.; Guo, Z.; Li, J.; Li, B.; Li, H.; and Lu, Y","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02392","last_updated":"2026-08-05T05:29:45Z","snapshot_observed_at":"2026-08-08T11:10:22.060960Z","submitted_at":"2026-08-03T15:35:28Z","title":"GROVE: Growing and Reasoning over Temporally Stratified Memory from Streaming Video Experience","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-04T08:12:47.458517Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2608.02392"},"observation_digest":"sha256:959eeb048b3af7611820f9a7eded69ec0961bd691f540209b69619824ab2c5b0","observation_id":"259dbeea-6a1e-41a5-89ad-a27f7e1e4aaf","resolution":{"observed_at":"2026-08-04T08:12:47.458517Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-08-07T00:14:15.929069Z","title":"Zhang, X.; Jia, Z.; Guo, Z.; Li, J.; Li, B.; Li, H.; and Lu, Y","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02392","last_updated":"2026-08-05T05:29:45Z","snapshot_observed_at":"2026-08-08T11:10:22.060960Z","submitted_at":"2026-08-03T15:35:28Z","title":"GROVE: Growing and Reasoning over Temporally Stratified Memory from Streaming Video Experience","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T00:14:15.929069Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2608.02392"},"observation_digest":"sha256:3f3ce84d52563c8fe4bcb670bf3de57d4dae2c537c56c5c28c912e89280f5b8b","observation_id":"499c13e1-ce69-46d3-89ec-54ff6b1c3754","resolution":{"observed_at":"2026-08-07T00:14:15.929069Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2406.08085/citation-record","integrity":"/paper/2406.08085/integrity","json":"/paper/2406.08085/citation-record.json","paper":"/paper/2406.08085"},"outbound":[],"paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-03T12:09:02.990658Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 50 inbound Pith citation observations for arXiv:2406.08085."}