{"as_of":"2026-08-15T13:24:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a0736bdf7d7838acc81c0f6b0e54f4073fd8b1d984ee785b874baa3a1421dbb3","coverage":[{"denominator":85,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":85,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T21:37:08.992465Z","state":"measured"},{"denominator":95,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":95,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-15T06:32:42.880941+00:00","state":"measured"},{"denominator":10,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":10,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-14T04:39:27.358904Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-10T12:15:01.137692Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.23825","snapshot_observed_at":"2026-08-06T00:03:35.208400Z","title":"Flash-vstream: Efficient real- time understanding for long video streams","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.04415","last_updated":"2025-08-06T13:03:16Z","snapshot_observed_at":"2026-08-07T20:42:58.746080Z","submitted_at":"2025-08-06T13:03:16Z","title":"Empowering Nanoscale Connectivity through Molecular Communication: A Case Study of Virus Infection","version":1},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-06T00:03:35.208400Z"},"links":{"cited_paper":"/paper/2506.23825","citing_paper":"/paper/2508.04415"},"observation_digest":"sha256:e00b7c61a7255a0bdc3024fe21d0e7628297bb4af4ea94ca567b763011b9542c","observation_id":"5d958aba-4e18-443b-bce7-01c2b8959940","resolution":{"observed_at":"2026-08-06T00:03:35.208400Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.23825","snapshot_observed_at":"2026-08-03T14:32:00.456065Z","title":"Flash-vstream: Efficient real- time understanding for long video streams.arXiv preprint arXiv:2506.23825, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2512.20117","last_updated":"2026-06-29T07:20:38Z","snapshot_observed_at":"2026-08-03T14:31:55.232086Z","submitted_at":"2025-12-23T07:21:21Z","title":"Delayed Bidirectional Alignment via Disentangled Audio Semantics for Audio-Visual Segmentation","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-03T14:32:00.456065Z"},"links":{"cited_paper":"/paper/2506.23825","citing_paper":"/paper/2512.20117"},"observation_digest":"sha256:48de8c0baf04b57c4104ad887e350c2d2c9b5fe008a75ac7e293bc928e8ebb6c","observation_id":"011ffc0d-af10-4df5-b428-909d0d7ea02e","resolution":{"observed_at":"2026-08-03T14:32:00.456065Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2506.23825","doi":"10.48550/arxiv.2506.23825","metadata_source":"arxiv_reference","pith_arxiv_id":"2506.23825","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"arXiv preprint arXiv:2506.23825 , year=","venue":null,"work_id":"d524fa9c-c052-4e82-bdda-e6a3955f763e","year":2025},"citing_paper":{"arxiv_id":"2604.10060","last_updated":"2026-04-11T06:54:56Z","snapshot_observed_at":"2026-08-10T21:22:44.566409Z","submitted_at":"2026-04-11T06:54:56Z","title":"Mosaic: Cross-Modal Clustering for Efficient Video Understanding","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-10T16:07:36.133404Z"},"links":{"cited_paper":"/paper/2506.23825","citing_paper":"/paper/2604.10060"},"observation_digest":"sha256:d677b8512cc5e4f16822c043289c610f5c0537d07036e70145d95718ca9812a4","observation_id":"6f3dff6f-4043-41f6-b592-79e133eaebbc","resolution":{"observed_at":"2026-05-10T16:10:34.388472Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2506.23825","doi":"10.48550/arxiv.2506.23825","metadata_source":"arxiv_reference","pith_arxiv_id":"2506.23825","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"arXiv preprint arXiv:2506.23825 , year=","venue":null,"work_id":"d524fa9c-c052-4e82-bdda-e6a3955f763e","year":2025},"citing_paper":{"arxiv_id":"2604.11627","last_updated":"2026-04-13T15:38:22Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-13T15:38:22Z","title":"POINTS-Long: Adaptive Dual-Mode Visual Reasoning in MLLMs","version":1},"reference_index":112,"source":"pdf_text","source_observed_at":"2026-05-10T15:23:08.671342Z"},"links":{"cited_paper":"/paper/2506.23825","citing_paper":"/paper/2604.11627"},"observation_digest":"sha256:4f6e42ea611d64bbd1cd1149cb0a2ddf330defa4300bfd93afbc57b3dade07d8","observation_id":"ce12786c-d594-42aa-8810-0d56971e0a6c","resolution":{"observed_at":"2026-05-11T10:41:03.709919Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2506.23825","doi":"10.48550/arxiv.2506.23825","metadata_source":"arxiv_reference","pith_arxiv_id":"2506.23825","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"arXiv preprint arXiv:2506.23825 , year=","venue":null,"work_id":"d524fa9c-c052-4e82-bdda-e6a3955f763e","year":2025},"citing_paper":{"arxiv_id":"2604.17052","last_updated":"2026-04-18T16:22:05Z","snapshot_observed_at":"2026-08-13T14:22:51.440674Z","submitted_at":"2026-04-18T16:22:05Z","title":"OASIS: On-Demand Hierarchical Event Memory for Streaming Video Reasoning","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-05-10T06:51:52.861981Z"},"links":{"cited_paper":"/paper/2506.23825","citing_paper":"/paper/2604.17052"},"observation_digest":"sha256:440d93ef279089c77a37469bc5b9b2ad367f280ffe6d933d9c1b7eb1a594d664","observation_id":"78041cb7-9628-4162-a2e3-c47795275266","resolution":{"observed_at":"2026-05-10T06:56:47.780231Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2506.23825","doi":"10.48550/arxiv.2506.23825","metadata_source":"arxiv_reference","pith_arxiv_id":"2506.23825","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"arXiv preprint arXiv:2506.23825 , year=","venue":null,"work_id":"d524fa9c-c052-4e82-bdda-e6a3955f763e","year":2025},"citing_paper":{"arxiv_id":"2605.07575","last_updated":"2026-05-11T11:58:53Z","snapshot_observed_at":"2026-07-06T23:19:56.415523Z","submitted_at":"2026-05-08T10:46:10Z","title":"Response-G1: Explicit Scene Graph Modeling for Proactive Streaming Video Understanding","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-05-11T02:30:55.939351Z"},"links":{"cited_paper":"/paper/2506.23825","citing_paper":"/paper/2605.07575"},"observation_digest":"sha256:a96ca0dc10f399906761edf966a30b021b5a9d3a85e2a6c984479cc03be3a262","observation_id":"44ceb307-4b36-4750-bc72-23d3b53ade81","resolution":{"observed_at":"2026-05-11T03:20:56.393925Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2506.23825","doi":"10.48550/arxiv.2506.23825","metadata_source":"arxiv_reference","pith_arxiv_id":"2506.23825","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"arXiv preprint arXiv:2506.23825 , year=","venue":null,"work_id":"d524fa9c-c052-4e82-bdda-e6a3955f763e","year":2025},"citing_paper":{"arxiv_id":"2605.07575","last_updated":"2026-05-11T11:58:53Z","snapshot_observed_at":"2026-07-06T23:19:56.415523Z","submitted_at":"2026-05-08T10:46:10Z","title":"Response-G1: Explicit Scene Graph Modeling for Proactive Streaming Video Understanding","version":2},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-05-12T03:00:34.728880Z"},"links":{"cited_paper":"/paper/2506.23825","citing_paper":"/paper/2605.07575"},"observation_digest":"sha256:bbad265f9de1f7f92bd9ca78fbf5c9ffc47f7fab8c167fb8e8668db44819f4c7","observation_id":"cf4b5ff5-efc3-4be8-997b-d36a18201cae","resolution":{"observed_at":"2026-05-12T03:01:17.733473Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2506.23825","doi":"10.48550/arxiv.2506.23825","metadata_source":"arxiv_reference","pith_arxiv_id":"2506.23825","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"arXiv preprint arXiv:2506.23825 , year=","venue":null,"work_id":"d524fa9c-c052-4e82-bdda-e6a3955f763e","year":2025},"citing_paper":{"arxiv_id":"2606.05917","last_updated":"2026-06-04T09:23:31Z","snapshot_observed_at":"2026-08-04T03:04:58.385559Z","submitted_at":"2026-06-04T09:23:31Z","title":"MemoryCard: Topic-Aware Multi-Modal Clue Compression for Long-Video Question Answering","version":1},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-06-28T01:52:33.768494Z"},"links":{"cited_paper":"/paper/2506.23825","citing_paper":"/paper/2606.05917"},"observation_digest":"sha256:2e61a7b61f977cca43744d1cdbf6c07d0b429dd05cb295469a2c572e310c6f50","observation_id":"54e3e141-b2f2-4945-be55-c0e5ffa8023f","resolution":{"observed_at":"2026-06-28T02:01:29.198189Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2506.23825","doi":"10.48550/arxiv.2506.23825","metadata_source":"arxiv_reference","pith_arxiv_id":"2506.23825","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"arXiv preprint arXiv:2506.23825 , year=","venue":null,"work_id":"d524fa9c-c052-4e82-bdda-e6a3955f763e","year":2025},"citing_paper":{"arxiv_id":"2606.08615","last_updated":"2026-06-07T13:00:19Z","snapshot_observed_at":"2026-08-05T22:03:12.300616Z","submitted_at":"2026-06-07T13:00:19Z","title":"Harnessing Streaming Video in the Wild","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-06-27T18:47:55.910417Z"},"links":{"cited_paper":"/paper/2506.23825","citing_paper":"/paper/2606.08615"},"observation_digest":"sha256:501d26f6a5806fa299bb976df2209b1e5782bab214e75732998f5fb75d2aa5aa","observation_id":"061c197b-c067-4d9a-bdf5-0e5971b2fd3a","resolution":{"observed_at":"2026-07-02T22:37:25.733342Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.23825","snapshot_observed_at":"2026-08-14T04:39:27.358904Z","title":"ICCV , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.08469","last_updated":"2026-08-09T04:22:57Z","snapshot_observed_at":"2026-08-15T00:10:53.359341Z","submitted_at":"2026-08-09T04:22:57Z","title":"Aero Realtime: Fully Aligned Input-Output Streams for Low-Latency Streaming Multimodal Generation","version":1},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-08-14T04:39:27.358904Z"},"links":{"cited_paper":"/paper/2506.23825","citing_paper":"/paper/2608.08469"},"observation_digest":"sha256:95c58c04a760d45de1ec79a34a1fb8e62d30b3ca4007e3ac9fdd64f6e9f237ca","observation_id":"e4615993-e9ff-4785-81b1-0e0ca671466b","resolution":{"observed_at":"2026-08-14T04:39:27.358904Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2506.23825/citation-record","integrity":"/paper/2506.23825/integrity","json":"/paper/2506.23825/citation-record.json","paper":"/paper/2506.23825"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-06T21:37:01.869344Z","title":"Gpt-4 technical report","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:01.869344Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:a5eb18838fd04b2b19ce2e1ff61fea08b20175a311873f3b50f88782177c36a1","observation_id":"6887147d-f18e-4ce0-be0b-649e9ee97769","resolution":{"observed_at":"2026-08-06T21:37:01.869344Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:01.948138Z","title":"Self-calibrated clip for training-free open-vocabulary segmentation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:01.948138Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:0f2ffd9250b1be65a6ced8123c09e636ac8abf5090d5c317829aeb48b14339fc","observation_id":"f951860e-258b-40dd-9b3e-695ac42e79ff","resolution":{"observed_at":"2026-08-06T21:37:01.948138Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:02.070349Z","title":"Memory consolidation enables long-context video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:02.070349Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:6e3627e86a0194a6a51e3da2a84433aebbbe0808b99ffd0dce6fd36e6002b7d0","observation_id":"e8c54936-2933-4df9-a915-0306862a540a","resolution":{"observed_at":"2026-08-06T21:37:02.070349Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:02.199304Z","title":"Language models are few-shot learners","venue":null,"work_id":null,"year":1901},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:02.199304Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:2d010ff4d4b63925f5deb72f431cac8e87990173da460aa9724460f63b691a6a","observation_id":"f1144657-4b06-4101-b70e-f10325e7f806","resolution":{"observed_at":"2026-08-06T21:37:02.199304Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1809.02105","last_updated":"2018-09-06T17:29:10Z","snapshot_observed_at":"2026-08-14T18:32:02.952068Z","submitted_at":"2018-09-06T17:29:10Z","title":"A Memory-Network Based Solution for Multivariate Time-Series Forecasting","version":1},"cited_work":{"arxiv_id":"1809.02105","doi":null,"metadata_source":"pith","pith_arxiv_id":"1809.02105","snapshot_observed_at":"2026-08-06T21:37:09.596774Z","title":"A Memory-Network Based Solution for Multivariate Time-Series Forecasting","venue":"cs.LG","work_id":"ae7c8f91-82e2-4bb0-b914-9ebeb0c0e06c","year":2018},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:02.378276Z"},"links":{"cited_paper":"/paper/1809.02105","citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:4240343d083643e034c1fd40d5379944f787317536c845adcdb054d0d11cce30","observation_id":"ed047457-67d2-493f-b1bd-4a5a5b54c0da","resolution":{"observed_at":"2026-08-06T21:37:09.602810Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:02.499259Z","title":"Distributed deep learning model for intelligent video surveillance systems with edge computing","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:02.499259Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:9ca356d8f13d4c458f5289176231105adc12b4169e4c8c1fff1c09e320901c0d","observation_id":"caa6d757-081e-424c-9949-076661df0e62","resolution":{"observed_at":"2026-08-06T21:37:02.499259Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:02.595350Z","title":"Videollm-online: Online video large language model for streaming video","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:02.595350Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:f8eb1359bd7655fd0003f0a099737409caf156175ce11ee2de6e80ccdd7a434c","observation_id":"aafc111a-b9fa-4326-966c-e495ab223121","resolution":{"observed_at":"2026-08-06T21:37:02.595350Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:02.735952Z","title":"Sharegpt4video: Improving video understanding and generation with better captions","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:02.735952Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:604ac099ea5fe5fb603027e9a8b9cc315f1b686c4835847d783abd858bd5f737","observation_id":"a3ad585c-95d6-4593-ae46-fc2760f3b7ac","resolution":{"observed_at":"2026-08-06T21:37:02.735952Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.715164Z","title":"Xmem: Long-term video object segmentation with an atkinson-shiffrin memory model","venue":null,"work_id":"df263340-41b6-4512-8df7-f7a390c9c57c","year":2022},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:02.844901Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:5d7944a79166a4499221b22024db11e84ea4adb5a91310e9858d180f386e3f83","observation_id":"7f9f6b6c-3478-4537-9fe5-05fe9a3a0293","resolution":{"observed_at":"2026-08-06T21:37:10.720770Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07476","last_updated":"2024-10-30T06:49:54Z","snapshot_observed_at":"2026-08-14T16:25:22.654846Z","submitted_at":"2024-06-11T17:22:23Z","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.07476","snapshot_observed_at":"2026-08-06T21:37:03.023513Z","title":"Videollama 2: Advancing spatial- temporal modeling and audio understanding in video-llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:03.023513Z"},"links":{"cited_paper":"/paper/2406.07476","citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:98b71e5b97ed4e9dee521d5b228ca21462122db87d22ecf6335190fe8b0ecea4","observation_id":"a8c5c172-dc6c-4e12-afaf-1b283327193c","resolution":{"observed_at":"2026-08-06T21:37:03.023513Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.699375Z","title":"Instructblip: Towards general-purpose vision-language models with instruction tuning","venue":null,"work_id":"09a41034-b46d-4b86-b68e-8b6895726a53","year":null},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:03.104975Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:5ecbb6f6a9abe8d2b0e7d96e42fddfb805b3183d74d789633e0a9931db623ad5","observation_id":"0fcbabcd-d5a3-40ab-b2f5-a01ba03c21a5","resolution":{"observed_at":"2026-08-06T21:37:10.704419Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.684215Z","title":"Flashattention-2: Faster attention with better paral- lelism and work partitioning","venue":null,"work_id":"0fccee00-3009-4e73-be08-5d2c73d2d583","year":2024},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:03.209972Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:4e71b1ffa3d252657f40f8b4a69fc285b81dd5e4b97b301c32a0da305086bb55","observation_id":"522ef2db-b67c-4aa7-924f-78c744c54477","resolution":{"observed_at":"2026-08-06T21:37:10.689071Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:03.307705Z","title":"An image is worth 16x16 words: Transformers for image recognition at scale","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:03.307705Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:de662db9c9db6400d5d883ac33d1d92f4753c1b4e7dca65590fb05a1dbe7edb5","observation_id":"f8ab067d-b2d0-4b68-ace1-74339480aca4","resolution":{"observed_at":"2026-08-06T21:37:03.307705Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-06T21:37:03.374979Z","title":"The llama 3 herd of models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:03.374979Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:fd01615596ade957b6ab38842e0774e6ca185e9cba5f81f4e566d2d68becfd1e","observation_id":"2f202760-4744-4c6a-aa31-8f2171aa0c05","resolution":{"observed_at":"2026-08-06T21:37:03.374979Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.658130Z","title":"A density-based algorithm for discovering clusters in large spatial databases with noise","venue":null,"work_id":"f62c595e-5796-4e71-ba15-fce59b2c5831","year":null},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:03.468755Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:66b31dbd42656d81d18e1b09fc6d3a4dcee76d5d3fbf1a4892ffb4d57d5f283a","observation_id":"53b89593-713c-4fb7-b545-9c19be088905","resolution":{"observed_at":"2026-08-06T21:37:10.662907Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.644283Z","title":"Video-mme: The first-ever comprehensive evaluation benchmark of multi-modal llms in video analysis","venue":null,"work_id":"e2e1325c-8f54-4ab5-babb-a2125b5d1df2","year":2025},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:03.534615Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:03ccee914e7ca505782f6c73ed23df327fb794d92f7537f9a3fedb32ade5a1de","observation_id":"f0d1da68-38ef-429f-8509-457ba7bebf19","resolution":{"observed_at":"2026-08-06T21:37:10.648958Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.630826Z","title":"Temporal sentence grounding in streaming videos","venue":null,"work_id":"a9ca1776-15a1-47cd-ad5c-8bc01c61d843","year":2023},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:03.687221Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:c69b57067785fb21b7dc2ab9378478f0ed47aa204f2c129770433024eaca4ae8","observation_id":"726d63d5-f4ff-4c22-bf44-8fd0f11dc009","resolution":{"observed_at":"2026-08-06T21:37:10.635328Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.616993Z","title":"Mist: Multi-modal iterative spatial- temporal transformer for long-form video question answering","venue":null,"work_id":"a5b791b9-d8f3-4bbb-8ec0-9a8c48e95262","year":2023},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:03.796093Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:ca2b5b9a5a818ebd0f33668cac73b74f7c9b4be69f81337d8230aad51f03f209","observation_id":"49749f40-590f-4bee-9800-da523fe23e36","resolution":{"observed_at":"2026-08-06T21:37:10.621253Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.602749Z","title":"Clip- adapter: Better vision-language models with feature adapters","venue":null,"work_id":"92447428-0485-4ffb-9381-95237e383bea","year":2024},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:03.874150Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:9efc19d60550b75d4a3b700b364b3b95c32f38f8578516f8281925b14ada64a2","observation_id":"f38e2cb4-da5b-433f-92b1-b18c187df245","resolution":{"observed_at":"2026-08-06T21:37:10.607482Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.589168Z","title":"Frameexit: Conditional early exiting for efficient video recognition","venue":null,"work_id":"f3c44d7c-2900-47fe-8322-cfbb78aac22d","year":2021},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:03.969979Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:558412a602ac64297b9a8ce245612200639ee0b3793da91d52c964b6c2c19781","observation_id":"a7e01be7-e4e9-49f0-822a-ac95500b5f06","resolution":{"observed_at":"2026-08-06T21:37:10.593252Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.576201Z","title":"Dynamic neural networks: A survey","venue":null,"work_id":"e4091326-36a6-4f8b-887f-2fdfbd5d7982","year":2021},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:04.043814Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:e85a870235cebe221637b2ef163e47b5d0dbef6fa7efc8b8524b9414d1525843","observation_id":"9b2576bf-59fc-4109-8f36-de56f08fe307","resolution":{"observed_at":"2026-08-06T21:37:10.580297Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.562909Z","title":"A twofold siamese network for real-time object tracking","venue":null,"work_id":"3c74e44a-fc66-4f04-b017-b885b91d4187","year":2018},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:04.091875Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:664e84dec8ae06f85b9b3ae446dc1bdc39fc3e9622c235e01b74bb943ec90aa6","observation_id":"908cd346-dabb-4b80-93c1-c82eb7e70c04","resolution":{"observed_at":"2026-08-06T21:37:10.566930Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.550113Z","title":"LoRA: Low-rank adaptation of large language models","venue":null,"work_id":"81f2bb1b-5154-490e-8144-b4ca86dd1d59","year":null},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:04.202477Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:ab1e6638c6fe9fa85ad7dcd7bed12577fa144f14803cfccc1e91322a7e0c3549","observation_id":"e0f959ed-ecc4-46cc-b449-3d955ef8b133","resolution":{"observed_at":"2026-08-06T21:37:10.554209Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.537207Z","title":"Movienet: A holistic dataset for movie understanding","venue":null,"work_id":"ecf606c8-03a4-4974-90a9-afce5fe6027f","year":2020},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:04.321669Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:b0e9b37d9908bc4fdb7304a1e359b471150257e226db8b910d7ce0cfb0eee590","observation_id":"c2a9f090-f48f-4e98-b09c-0c5cd7f447a8","resolution":{"observed_at":"2026-08-06T21:37:10.541398Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.522040Z","title":"Chat-univi: Unified visual representation em- powers large language models with image and video under- standing","venue":null,"work_id":"eedd0201-7513-4604-a164-9014753694c3","year":2024},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:04.437980Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:3ad8369d25e5441471dc0ffad6ba3c6f92b4ae57230349e4c894b2e1e7441608","observation_id":"461b260e-ed3d-44a6-9ecc-b9003b49afea","resolution":{"observed_at":"2026-08-06T21:37:10.527716Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.508466Z","title":"Seed-bench: Benchmarking multimodal large language models","venue":null,"work_id":"72626e7a-9bca-48dc-badf-b468f2736f02","year":2024},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:04.531283Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:d3a0f10bebf4e6ee13df3f6d57c7cf538e959a15c73c084a51a78d2ca2269bf0","observation_id":"20d5ba49-83be-4a5b-9471-84ffaab9a37b","resolution":{"observed_at":"2026-08-06T21:37:10.512597Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03326","last_updated":"2024-10-26T16:35:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:59:44Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03326","snapshot_observed_at":"2026-08-06T21:37:04.625398Z","title":"Llava-onevision: Easy visual task transfer","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:04.625398Z"},"links":{"cited_paper":"/paper/2408.03326","citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:dc6facfda1a0a5241eab646813530f4eaddbda4e56bba24dd7561d786878d0b8","observation_id":"2bb99f58-390d-4cf3-a8b6-fd826d376847","resolution":{"observed_at":"2026-08-06T21:37:04.625398Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.494160Z","title":"Blip: Bootstrapping language-image pre-training for unified vision- language understanding and generation","venue":null,"work_id":"145301db-bd3e-43b1-b28b-59a5ab6e7b67","year":2022},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:04.695915Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:7f58a9ab33360753ab0defae7e259f2ab74084ef035ec7a16f7517487be82833","observation_id":"c647090a-7381-4c06-bed0-c3a1c5d8f60c","resolution":{"observed_at":"2026-08-06T21:37:10.498813Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.479406Z","title":"Blip- 2: Bootstrapping language-image pre-training with frozen image encoders and large language models","venue":null,"work_id":"c6d95875-26b0-4dec-953c-2c64d9faa0b8","year":2023},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:04.769774Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:981eb32ab23e207aa7dfc2847a75625241c4ed9ada6e9dc15610f28d14e9dc45","observation_id":"391d44d2-f142-4fd4-9d13-ea37546e6ca8","resolution":{"observed_at":"2026-08-06T21:37:10.484102Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.465792Z","title":"Mvbench: A comprehensive multi-modal video understand- ing benchmark","venue":null,"work_id":"bab082ca-5b0e-42d2-9618-25920eee08d5","year":2024},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:04.874604Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:0ee3439846b937243187752e86805501f75fccce5df9d919d89df45afe285c90","observation_id":"d3073307-a94a-44a6-b442-c23c61166907","resolution":{"observed_at":"2026-08-06T21:37:10.470259Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.452777Z","title":"Llama-vid: An image is worth 2 tokens in large language models","venue":null,"work_id":"61a09c4c-5bdc-4033-ad5c-d5465479e706","year":2025},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:05.041607Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:316b7e5e07682a55d23ec3e975918b9807bc899dea764e6ec33dacc08dde5088","observation_id":"82c186d8-45f2-428d-b58a-35d1f51e6d15","resolution":{"observed_at":"2026-08-06T21:37:10.456649Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.440247Z","title":"Visual instruction tuning","venue":null,"work_id":"8e02745f-4938-4f6c-88de-ebf0228f8c22","year":2023},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:05.162198Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:4999c9c685965aa07b403c4c7d0e713942cad9ea50f04d79ac2f30ba3f9cc09f","observation_id":"17555231-f16f-4331-b585-71124f0408f2","resolution":{"observed_at":"2026-08-06T21:37:10.444108Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.425743Z","title":"Improved baselines with visual instruction tuning","venue":null,"work_id":"55e63e5e-4414-4a5a-b00c-cb2d8afebef3","year":2024},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:05.248330Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:4e0ef85554c13fd41da49bf6ecc3ff2ce48dd0a42c3bad30f209680b87b20a4b","observation_id":"1d17eb93-f65c-41fa-9216-1e97d9c997c0","resolution":{"observed_at":"2026-08-06T21:37:10.430583Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-08-13T04:53:28.750571Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-08-06T21:37:05.338028Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:05.338028Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:30a244f33f0e99c29ade07b0a4ee209dbac25deb273463c789659da3f38b11c5","observation_id":"0f4f6773-ee34-41bd-bd60-e24713ee2a90","resolution":{"observed_at":"2026-08-06T21:37:05.338028Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.410888Z","title":"Learning quality-aware dynamic mem- ory for video object segmentation","venue":null,"work_id":"3fbde522-2535-4424-b809-209cb6242503","year":null},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:05.434941Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:8b8224780cff434ec3012fba79c2a6624b9e2e447898862a575668f80d529923","observation_id":"fe3eef6e-9ba6-4747-9420-16d2f1d9c9b1","resolution":{"observed_at":"2026-08-06T21:37:10.415628Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.397664Z","title":"Universal segmentation at arbitrary granularity with language instruction","venue":null,"work_id":"f024f518-a15f-47ed-b0af-f6da4aa1d1d3","year":2024},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:05.607752Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:748d4225dcafd4264f687861956e941f787ac234027ae7fdc989ddaea8a6176e","observation_id":"181b243a-8ba6-427b-a710-d3b95a117f09","resolution":{"observed_at":"2026-08-06T21:37:10.401854Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12961","last_updated":"2025-02-27T06:09:46Z","snapshot_observed_at":"2026-08-12T22:41:32.229292Z","submitted_at":"2024-09-19T17:59:51Z","title":"Oryx MLLM: On-Demand Spatial-Temporal Understanding at Arbitrary Resolution","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12961","snapshot_observed_at":"2026-08-06T21:37:05.734732Z","title":"Oryx mllm: On-demand spatial- temporal understanding at arbitrary resolution","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:05.734732Z"},"links":{"cited_paper":"/paper/2409.12961","citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:a49e36617a003c1c33a242332645b03ea68b6ca8e455b3d09894efae26d9ca16","observation_id":"9a7864cc-bbaa-41c0-bd7b-dca40e139fa4","resolution":{"observed_at":"2026-08-06T21:37:05.734732Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.384456Z","title":"Soc: Semantic- assisted object cluster for referring video object segmentation","venue":null,"work_id":"bf552756-b484-4717-9643-bda3c8110cdf","year":2023},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:05.827409Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:32936053c655eeb74eb1e4d496e5b553ca1986f6104177a7f40546e96f5834dc","observation_id":"cc6f9ea1-d404-4abd-b1e3-0c16b116cd10","resolution":{"observed_at":"2026-08-06T21:37:10.388612Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.371149Z","title":"Multi-task deep learning for real-time 3d human pose estimation and action recognition","venue":null,"work_id":"604fb36a-050d-4d5b-ac79-eb8ad6abe62f","year":2020},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:05.934945Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:5d64b958de6158ffe77d632a7e55dba8c23a5f32d4be279bf2d883f0ba02a2a7","observation_id":"8f68580d-5ba6-40b9-a3d8-7caff45da478","resolution":{"observed_at":"2026-08-06T21:37:10.375454Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.357843Z","title":"Vista-llama: Reducing hallucination in video language models via equal distance to visual tokens","venue":null,"work_id":"28381253-c945-4d9c-a849-30fcafaaf86f","year":2024},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:06.023144Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:f9d6d7ca03e7cd852965120eceece06a75bc28fd8bc92eb21e2367f2ba775a6f","observation_id":"22555c9a-deb7-4a70-8ce1-2f4a2a885b9d","resolution":{"observed_at":"2026-08-06T21:37:10.362008Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.344249Z","title":"Video-chatgpt: Towards detailed video under- standing via large vision and language models","venue":null,"work_id":"c3da141b-94c4-424f-809b-e0008a00584d","year":2024},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:06.087437Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:39b5d1a45e038aa79e7fd5a62fee4651b9b45fcb500ec943b4ec567865f80e76","observation_id":"be611fb7-0e21-471b-8211-369fb2e96bde","resolution":{"observed_at":"2026-08-06T21:37:10.348389Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.330492Z","title":"Some methods for classification and anal- ysis of multivariate observations","venue":null,"work_id":"0cc50744-8584-4e3e-aedf-437165abc544","year":1967},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:06.189654Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:6c669c4bc354cb3e1ebbd1d7d734640fa73937080afdeb16b2ca300f52e2b4a3","observation_id":"41fb913b-96b3-4fec-9d33-77a80d89d812","resolution":{"observed_at":"2026-08-06T21:37:10.334609Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.315975Z","title":"Egoschema: A diagnostic benchmark for very long- form video language understanding","venue":null,"work_id":"7f9faad8-ef0a-48f9-ac31-fa0c93f4270b","year":2023},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:06.280529Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:5a33fffd467c832f9ba952566a5e466d303ebdb343d38e778db89b7ee31ad130","observation_id":"7f33954e-d803-4a93-a719-e53fc98c3706","resolution":{"observed_at":"2026-08-06T21:37:10.320839Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.301705Z","title":"Deepres: A deep learning-based video summarization strategy for resource- constrained industrial surveillance scenarios","venue":null,"work_id":"af52da88-279d-4c41-9000-48966360c5e3","year":2019},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:06.351008Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:2ae24cf3b7699be042b290233efb5e8e874a1b7a29f4cba621117736de7082fd","observation_id":"ba0da123-c123-4aa5-b457-02ca53d8da66","resolution":{"observed_at":"2026-08-06T21:37:10.306067Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.287121Z","title":"Training language models to follow instructions with human feedback","venue":null,"work_id":"e9a72199-566f-42ac-84f5-5d91529f1ebf","year":2022},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:06.424661Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:6181262af53de87820042545a8c5a31d375de3cb70d49a30f78d53622c62b694","observation_id":"c70825d6-5f2e-4b0d-948d-316be74c920a","resolution":{"observed_at":"2026-08-06T21:37:10.291436Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.272541Z","title":"Streaming long video understanding with large language models","venue":null,"work_id":"d6e2af2d-b7b2-412d-a9ec-21732ab67153","year":2025},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:06.479104Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:a5e4316bcb0d19de54572ca7cbf3b0c0822f8acc1a842ee38006cfd32d72767d","observation_id":"a26de3d5-160b-42bd-b8b9-9b830f97177c","resolution":{"observed_at":"2026-08-06T21:37:10.276943Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.258315Z","title":"Timechat: A time-sensitive multimodal large language model for long video understanding","venue":null,"work_id":"2ba1b5bf-2086-4994-863b-035066fd7b31","year":null},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:06.539428Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:3af302fb58c42a04e415e0a520105fb7864d726d4d931f6655c806bda47a488b","observation_id":"4b17bef0-607e-4494-8825-a72a1012c341","resolution":{"observed_at":"2026-08-06T21:37:10.263120Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.243820Z","title":"Robovqa: Multimodal long-horizon reasoning for robotics","venue":null,"work_id":"80f376f8-ef10-4629-beca-d1e39bb4db7f","year":2024},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:06.601443Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:e434a548e6aaad5346af9b5e0c7181cf61fb8de296d8840a6c047eb332e8f229","observation_id":"f2a3b187-e579-4d79-abc2-17de6c4e4e55","resolution":{"observed_at":"2026-08-06T21:37:10.248033Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.227071Z","title":"Moviechat: From dense token to sparse memory for long video understanding","venue":null,"work_id":"33f8b17d-1ab4-4e59-9bae-183263b4e444","year":2024},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:06.651780Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:f9e482fa98b0633c976d083708f6343135c55206ac2cf67cc307b4c8a5382d9e","observation_id":"964afe38-b9f3-4395-b522-4bdd7d6198ae","resolution":{"observed_at":"2026-08-06T21:37:10.232186Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:06.703267Z","title":"Roformer: Enhanced transformer with rotary position embedding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:06.703267Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:288cff09d094e651ff1c9bf13c1a0c759556950af8b7e38729af7872161df732","observation_id":"53fc9ae4-2eb8-4020-acdd-603aa9cd8429","resolution":{"observed_at":"2026-08-06T21:37:06.703267Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.197515Z","title":"Tracking as online decision-making: Learning a policy from streaming videos with reinforcement learning","venue":null,"work_id":"2d84a1de-d803-4b54-939b-ec85833badf9","year":2017},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:06.822196Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:8021929b693677b18c0ca0407610f70235883c62e4083d63fb7039ef2111fd4e","observation_id":"e2563ac0-4c65-4138-859c-d37055ffaebf","resolution":{"observed_at":"2026-08-06T21:37:10.202961Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.174946Z","title":"Dynamic memory based attention network for sequential recommendation","venue":null,"work_id":"d28396a5-1188-4ef3-8285-991a1be7099b","year":2021},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:06.916665Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:36da1a920275d11ebe707251ded7464e1026b3bf6c88f8ccaa13562a0480d045","observation_id":"cfcecbf7-a784-46a9-b8e0-7978fc80d6fd","resolution":{"observed_at":"2026-08-06T21:37:10.180383Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05530","last_updated":"2024-12-16T17:39:39Z","snapshot_observed_at":"2026-08-14T18:15:53.516440Z","submitted_at":"2024-03-08T18:54:20Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05530","snapshot_observed_at":"2026-08-06T21:37:07.008868Z","title":"Gemini 1.5: Unlocking mul- timodal understanding across millions of tokens of context","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:07.008868Z"},"links":{"cited_paper":"/paper/2403.05530","citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:2fdad985c291d518f017b7a3f73f0a80cf730b90e21e50f1673a304ddce0d7ed","observation_id":"6125cca0-ee24-438a-a6f3-d2b3b76defe7","resolution":{"observed_at":"2026-08-06T21:37:07.008868Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.07491","last_updated":"2025-06-23T13:45:50Z","snapshot_observed_at":"2026-08-14T19:17:32.674749Z","submitted_at":"2025-04-10T06:48:26Z","title":"Kimi-VL Technical Report","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.07491","snapshot_observed_at":"2026-08-06T21:37:07.074611Z","title":"Kimi-vl technical report","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:07.074611Z"},"links":{"cited_paper":"/paper/2504.07491","citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:cb8accb89e50e3afca3014463f36cab3b897f91284e8095c323fc360e359718f","observation_id":"b37c7254-c536-42ff-82bb-7afa328b9445","resolution":{"observed_at":"2026-08-06T21:37:07.074611Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-06T21:37:07.127746Z","title":"Llama: Open and efficient foundation language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:07.127746Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:a79137316bc89039e286d9c8753b595b0aed111789687651ef844d025b2985b1","observation_id":"cb49f620-fa9a-4472-9dfe-870c0cb1ac04","resolution":{"observed_at":"2026-08-06T21:37:07.127746Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-06T21:37:07.188899Z","title":"Llama 2: Open foundation and fine-tuned chat models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:07.188899Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:95591efe86d75b6dc9c3691e57c1c5d45855a476787ed61a3da7821f61de1b46","observation_id":"87ecba3e-14eb-4e4e-a8ff-8721d5970736","resolution":{"observed_at":"2026-08-06T21:37:07.188899Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12191","last_updated":"2024-10-03T15:54:49Z","snapshot_observed_at":"2026-08-06T05:35:29.109022Z","submitted_at":"2024-09-18T17:59:32Z","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12191","snapshot_observed_at":"2026-08-06T21:37:07.239662Z","title":"Qwen2-vl: Enhancing vision-language model’s perception of the world at any resolution","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:07.239662Z"},"links":{"cited_paper":"/paper/2409.12191","citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:b008e92312d3d58584e07c67ef02c302cf84da50d0ca9a515defb9ccdfb5f58b","observation_id":"a350224f-da1f-4b50-90cf-84aaac305b3e","resolution":{"observed_at":"2026-08-06T21:37:07.239662Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08035","last_updated":"2025-08-09T10:54:59Z","snapshot_observed_at":"2026-08-08T19:38:26.415599Z","submitted_at":"2024-06-12T09:36:52Z","title":"LVBench: An Extreme Long Video Understanding Benchmark","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08035","snapshot_observed_at":"2026-08-06T21:37:07.288161Z","title":"Lvbench: An extreme long video understanding benchmark","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:07.288161Z"},"links":{"cited_paper":"/paper/2406.08035","citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:92b2415c94e5a20a26fac06559b8aa28e512f73845dea56a9514dfadf7024442","observation_id":"77574e63-3d4e-4b9b-adf3-d410f4e6881a","resolution":{"observed_at":"2026-08-06T21:37:07.288161Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.20504","last_updated":"2025-03-24T02:17:34Z","snapshot_observed_at":"2026-08-10T23:17:19.220083Z","submitted_at":"2024-12-29T15:42:24Z","title":"ReTaKe: Reducing Temporal and Knowledge Redundancy for Long Video Understanding","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.20504","snapshot_observed_at":"2026-08-06T21:37:07.383398Z","title":"Retake: Reducing temporal and knowledge redundancy for long video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:07.383398Z"},"links":{"cited_paper":"/paper/2412.20504","citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:272a8c13c61f5b0fefd31325ba43ecca5c7abbc28a4187c977c334c62228bbf8","observation_id":"242fa570-3d84-43c4-8f27-c01f37a67a4e","resolution":{"observed_at":"2026-08-06T21:37:07.383398Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.157757Z","title":"Adaptive focus for efficient video recognition","venue":null,"work_id":"ee10e6bf-208b-4db8-85ae-de6d877a4704","year":2021},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:07.498070Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:8b00cd6684c13115fa3781a32f3bc990804986386538007a504fa5c704de151d","observation_id":"5f9bbaf5-76ae-4e83-8855-cd7253172285","resolution":{"observed_at":"2026-08-06T21:37:10.162546Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.140762Z","title":"Adafocus v2: End-to-end training of spatial dynamic net- works for video recognition","venue":null,"work_id":"28a36a6b-c83e-4ca1-a35a-89fd8b719466","year":2022},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:07.576870Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:bbd6e79ed788f3d9230cd80697dc158a7ce04cfa7dc10b0284025c20916ee25a","observation_id":"f9a7a55a-0646-40da-a775-58ee96e5faa1","resolution":{"observed_at":"2026-08-06T21:37:10.146070Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.119865Z","title":"Adafocusv3: On unified spatial-temporal dynamic video recognition","venue":null,"work_id":"48be9336-d38f-48ee-8fab-15d4372d072e","year":2022},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:07.596032Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:bafd95b8c641491bbc1ec24df8cbcda70cf91e409e8d72f93fec3506c780f84a","observation_id":"81c0ee5e-a38e-419d-a8b0-f15e86e45ecf","resolution":{"observed_at":"2026-08-06T21:37:10.124808Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00603","last_updated":"2024-12-15T12:04:12Z","snapshot_observed_at":"2026-08-12T23:32:00.417126Z","submitted_at":"2024-06-30T06:08:12Z","title":"Hierarchical Memory for Long Video QA","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.00603","snapshot_observed_at":"2026-08-06T21:37:07.699169Z","title":"Hierarchical memory for long video qa","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:07.699169Z"},"links":{"cited_paper":"/paper/2407.00603","citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:b4d8fa10ffcf68d95dca180225501e13f37c54256a935255b885683d068491f7","observation_id":"bbdda88f-2070-44df-9029-f53700619c6f","resolution":{"observed_at":"2026-08-06T21:37:07.699169Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.01268","last_updated":"2024-12-02T08:35:31Z","snapshot_observed_at":"2026-08-13T16:29:43.780042Z","submitted_at":"2024-12-02T08:35:31Z","title":"Ponder & Press: Advancing Visual GUI Agent towards General Computer Control","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.01268","snapshot_observed_at":"2026-08-06T21:37:07.848372Z","title":"Ponder & press: Advancing visual gui agent towards general computer control","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:07.848372Z"},"links":{"cited_paper":"/paper/2412.01268","citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:e7997fd5eab5e6e47af31436c80a71204790c253a8cb84a6f0d6e17e0824fd09","observation_id":"7c31022a-e811-44a1-adfe-b975d579c6d7","resolution":{"observed_at":"2026-08-06T21:37:07.848372Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.105440Z","title":"Uni-adafocus: Spatial-temporal dynamic computation for video recognition","venue":null,"work_id":"119c9729-43ca-42fb-91e6-3cb5ae0f1baf","year":2024},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:08.015864Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:8ab1075991fd1f21c1b6563234ba215cc4d6bbf252813dcca27c8a2a8aeeb68a","observation_id":"983783b2-0c3d-4687-88f8-c2bab012ce25","resolution":{"observed_at":"2026-08-06T21:37:10.110255Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.091855Z","title":"Iterprime: Zero-shot referring image segmentation with iterative grad-cam refinement and primary word emphasis","venue":null,"work_id":"42c85def-14cb-47b6-9ffc-c741a3811573","year":2025},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:08.185558Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:9be68226fdbc50a200387c2b9ea5353a74521abf4432afa15b533932c088e0a7","observation_id":"05e41abc-eb7f-4d65-90f7-facf52cbcefb","resolution":{"observed_at":"2026-08-06T21:37:10.096179Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.077152Z","title":"Sam2-love: Segment anything model 2 in language- aided audio-visual scenes","venue":null,"work_id":"7edf9e6a-1014-4d6a-a458-321f6a43b519","year":null},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:08.352293Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:32da5131e14f52891939d8ea28f43d0733cf09c36f58982aedbdc89c0c597d8d","observation_id":"108b516f-0500-4305-a3ed-f0802abfa1fa","resolution":{"observed_at":"2026-08-06T21:37:10.081701Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.060253Z","title":"Towards real-time multi-object tracking","venue":null,"work_id":"fe15232c-0aef-4c15-af37-9d8bc76d424e","year":2020},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:08.520067Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:5da343f88dfeea035c9b163adce1da18e62f847c813034420ebc4cd4d33c59ba","observation_id":"9f00587a-b7bb-4c24-812d-f4a16f01545a","resolution":{"observed_at":"2026-08-06T21:37:10.066009Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.041173Z","title":"Videollm-mod: Efficient video- language streaming with mixture-of-depths vision compu- tation","venue":null,"work_id":"c7805d39-4c68-4e99-b0d0-75fce801a22b","year":2024},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:08.686159Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:1d84925bc60dd8208160d1303a1ea2813f56ed5785ce6b661bbc05f2aa2fdaa6","observation_id":"66990da4-9ee6-4cdd-894d-9844412dd7d1","resolution":{"observed_at":"2026-08-06T21:37:10.048899Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.024234Z","title":"Next-qa: Next phase of question-answering to explaining temporal actions","venue":null,"work_id":"6cc52b04-5db4-4ec1-9108-41f980c01f38","year":2021},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:08.802065Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:9262df57ff109e35bceb7c8cf5dec975b531c3cc87b48039684c476f99725fe7","observation_id":"f474a2e7-700c-40c7-beab-a09b604054f4","resolution":{"observed_at":"2026-08-06T21:37:10.029636Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.10188","last_updated":"2024-12-13T02:32:06Z","snapshot_observed_at":"2026-08-05T14:57:53.592979Z","submitted_at":"2024-08-19T17:48:08Z","title":"LongVILA: Scaling Long-Context Visual Language Models for Long Videos","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.10188","snapshot_observed_at":"2026-08-06T21:37:08.873755Z","title":"Longvila: Scaling long-context visual language models for long videos","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:08.873755Z"},"links":{"cited_paper":"/paper/2408.10188","citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:b4766d4f5ca450ea3945084b9c6479b3984e2ad5addacd2bcc25fbf25c31c29a","observation_id":"93c29c6b-51ce-4a07-9a67-99f57cf17501","resolution":{"observed_at":"2026-08-06T21:37:08.873755Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:10.007260Z","title":"Fine-grained video captioning via graph-based multi- granularity interaction learning","venue":null,"work_id":"f8e3d68b-b115-4218-8fa7-da0109032afe","year":2019},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:08.879930Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:f0e0bbdace954ab9507d074363c6b033479cef2c693c00650a70d780e9ccfa55","observation_id":"9eb08b61-673f-44b1-8016-d638a7fe5d3d","resolution":{"observed_at":"2026-08-06T21:37:10.012337Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10671","last_updated":"2024-09-10T13:25:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-07-15T12:35:42Z","title":"Qwen2 Technical Report","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.10671","snapshot_observed_at":"2026-08-06T21:37:08.890274Z","title":"Qwen2 technical report","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:08.890274Z"},"links":{"cited_paper":"/paper/2407.10671","citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:3e1d6526b55570cdd049a3a159b674e233829559bf40905161239532e68875b9","observation_id":"b050451d-f53c-491f-950b-74967d66ffce","resolution":{"observed_at":"2026-08-06T21:37:08.890274Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:09.987002Z","title":"Language-aware vision transformer for referring segmentation","venue":null,"work_id":"ae65eae8-3663-4410-80ce-27b7c94739d2","year":2024},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:08.900750Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:9d59e70ea67f78ffcd8496dd9d68f4f31c0a0ed0c20e5779f25344fd10b46fdf","observation_id":"fbe6b9ba-319a-4d92-9783-f9ee5860840b","resolution":{"observed_at":"2026-08-06T21:37:09.994052Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:09.970164Z","title":"Atp-llava: Adaptive token pruning for large vision language models","venue":null,"work_id":"6d869d7a-3880-4bb2-9e84-8c2375b12380","year":2025},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:08.917306Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:763b10e36f8c8d8b1467f80c1d33a09672ced736d1c82ff150bdb06965f76e01","observation_id":"1cf98fa4-906c-44af-af7c-e0594050c865","resolution":{"observed_at":"2026-08-06T21:37:09.975487Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:09.952029Z","title":"V oco-llama: Towards vision compression with large language models","venue":null,"work_id":"5d863902-d6b3-4515-b234-19770f4da034","year":null},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:08.923634Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:6236f43f4f1af0af537791716d34fedfb5fbc6ae073186054f7d5d62907dcae1","observation_id":"d93939c1-ba4a-49c8-bdbf-69ed035ef2a7","resolution":{"observed_at":"2026-08-06T21:37:09.958375Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:09.926841Z","title":"Self-chained image-language model for video localization and question answering","venue":null,"work_id":"2fb3993d-5f63-462d-8d1e-daf7a0e3a1de","year":2023},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:08.934527Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:b67766c0befefe348a99efb503a62c4780dc739bbdf17a35d9a2f88585a3fb83","observation_id":"f40af02a-f349-43ba-b593-e05875441c11","resolution":{"observed_at":"2026-08-06T21:37:09.933447Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:09.905311Z","title":"Activitynet-qa: A dataset for understanding complex web videos via question answering","venue":null,"work_id":"e54622dc-2a70-41cc-a77c-4dc8dbbdc8bf","year":2019},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:08.942219Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:cf57470d92d996539655bb952186c4099ba50e36b748909c2c9b6673e5cd9f96","observation_id":"5708aa3f-d747-4b72-9ae2-314e41a9b0a1","resolution":{"observed_at":"2026-08-06T21:37:09.913146Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:09.887409Z","title":"Real-time action recognition with enhanced motion vector cnns","venue":null,"work_id":"9571a443-80fa-4bf2-bb14-fecc3d1fba7a","year":2016},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:08.956859Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:3663235dfde5635f75c68e54f241f9c0b570bd1fe48224f22eb241db4fd4f771","observation_id":"3b095665-9420-4c13-8203-70c5b0f8f2b4","resolution":{"observed_at":"2026-08-06T21:37:09.892300Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.16852","last_updated":"2024-07-01T02:59:29Z","snapshot_observed_at":"2026-08-07T09:52:45.942315Z","submitted_at":"2024-06-24T17:58:06Z","title":"Long Context Transfer from Language to Vision","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.16852","snapshot_observed_at":"2026-08-06T21:37:08.962828Z","title":"Long context transfer from language to vision","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:08.962828Z"},"links":{"cited_paper":"/paper/2406.16852","citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:7bcccfb5dd2a31a7ba01abce1bd2576387657b07df49b5a6bde6ba5559fab682","observation_id":"fafdc095-d006-47ba-8474-889537b0e4ff","resolution":{"observed_at":"2026-08-06T21:37:08.962828Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-06T21:37:08.967515Z","title":"Video instruction tuning with synthetic data","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:08.967515Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:acdb4cba53e24eba0d10d687ac1bb446b2baaea79feb704269aa946e91aed364","observation_id":"113072f7-b149-48b4-98a8-8e9f4e1254e5","resolution":{"observed_at":"2026-08-06T21:37:08.967515Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04264","last_updated":"2025-01-01T15:53:58Z","snapshot_observed_at":"2026-08-03T20:38:36.602554Z","submitted_at":"2024-06-06T17:09:32Z","title":"MLVU: Benchmarking Multi-task Long Video Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04264","snapshot_observed_at":"2026-08-06T21:37:08.972286Z","title":"Mlvu: A comprehensive benchmark for multi-task long video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:08.972286Z"},"links":{"cited_paper":"/paper/2406.04264","citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:5f85de197d8d704eece507c3a22cba7c69cebfea1e98c1da1d2b3996d45eb628","observation_id":"52e61739-f8b5-43b7-a14c-62e6067bddfc","resolution":{"observed_at":"2026-08-06T21:37:08.972286Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:09.870064Z","title":"Streaming dense video captioning","venue":null,"work_id":"d54221fe-bb13-468c-bbcf-eb6587206f75","year":2024},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:08.977508Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:f88736664a69e0adf531c1998ab902f5c0ed56f0f5c920fe461b057a9ed3dc1f","observation_id":"2c8153d6-b46f-4322-b8f9-89626a8f23b0","resolution":{"observed_at":"2026-08-06T21:37:09.875831Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.15513","last_updated":"2025-04-22T01:19:53Z","snapshot_observed_at":"2026-08-14T13:40:28.599931Z","submitted_at":"2025-04-22T01:19:53Z","title":"InstaRevive: One-Step Image Enhancement via Dynamic Score Matching","version":1},"cited_work":{"arxiv_id":"2504.15513","doi":null,"metadata_source":"pith","pith_arxiv_id":"2504.15513","snapshot_observed_at":"2026-08-06T21:37:09.046163Z","title":"InstaRevive: One-Step Image Enhancement via Dynamic Score Matching","venue":"cs.CV","work_id":"9ea79823-a8cb-4ffc-868a-51f1ebb66880","year":2025},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:08.984190Z"},"links":{"cited_paper":"/paper/2504.15513","citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:f2de6e2a9bb47beaae63029f2b7a25921e1ff9f7d77d8728ad8fa0fff6fa6a97","observation_id":"def39505-06cc-4ccf-baed-4dff0a300a59","resolution":{"observed_at":"2026-08-06T21:37:09.053145Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:37:09.850374Z","title":null,"venue":null,"work_id":"60998ed8-c1cf-425d-a460-e1a562020756","year":1920},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":1000,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:08.992465Z"},"links":{"citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:18b2f0824cca21d66a6b823bdda4f4ca6e33d37142c05cd0424de3bf074fd128","observation_id":"e10025c3-8e4e-4ce7-8f23-a47b69e63cce","resolution":{"observed_at":"2026-08-06T21:37:09.856397Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-14T13:48:33.004337Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams"},"reference_resolution":{"displayed":85,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":29,"verified_exact":2,"verified_fuzzy":54},"total_outbound_references":85},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"thesis":"As of 15 August 2026, this Paper Citation Record lists 85 of 85 outbound references and 10 inbound Pith citation observations for arXiv:2506.23825."}