{"as_of":"2026-08-19T21:28:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:f3d380a51687fcca58d030d603d83c74738b72732b0a654ab282532c2bc194aa","coverage":[{"denominator":19,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":19,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-10T14:47:54.917408Z","state":"measured"},{"denominator":20,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":20,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-19T06:32:44.657259+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-01T12:28:00.885137Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-08-14T05:50:11.493950Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.01858","snapshot_observed_at":"2026-08-01T12:28:00.885137Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.19547","last_updated":"2026-07-21T19:46:38Z","snapshot_observed_at":"2026-08-15T03:02:05.250212Z","submitted_at":"2026-07-21T19:46:38Z","title":"ChronoStitch: Training-Free Composition of Visual KV Memories for Long-Horizon Temporal Reasoning","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-01T12:28:00.885137Z"},"links":{"cited_paper":"/paper/2605.01858","citing_paper":"/paper/2607.19547"},"observation_digest":"sha256:acd0b971a455a07c00f51e6afa280504b84198a34f357d702260fa921c027e4a","observation_id":"9a879610-a4d1-493d-9ec5-98b5baa78904","resolution":{"observed_at":"2026-08-01T12:28:00.885137Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2605.01858/citation-record","integrity":"/paper/2605.01858/integrity","json":"/paper/2605.01858/citation-record.json","paper":"/paper/2605.01858"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-08-14T04:17:22.593941Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":"2502.13923","doi":"10.48550/arxiv.2502.13923","metadata_source":"pith","pith_arxiv_id":"2502.13923","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen2.5-VL Technical Report","venue":"cs.CV","work_id":"69dffacb-bfe8-442d-be86-48624c60426f","year":2025},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-08-14T05:50:11.493950Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"cited_paper":"/paper/2502.13923","citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:4b6de018eaa49aa10d9823733ea5aec99e5825619197ec8279d05aa03959859b","observation_id":"27b29a5d-81b8-479e-8e4a-c7c3431e3540","resolution":{"observed_at":"2026-05-11T11:31:03.544517Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-08T16:08:16.864468+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T16:08:16.864468+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.13915","last_updated":"2025-04-10T17:13:08Z","snapshot_observed_at":"2026-08-16T12:42:17.592988Z","submitted_at":"2025-04-10T17:13:08Z","title":"Memory-efficient Streaming VideoLLMs for Real-time Procedural Video Understanding","version":1},"cited_work":{"arxiv_id":"2504.13915","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.13915","snapshot_observed_at":"2026-07-04T13:19:50.790416Z","title":"Memory-efficient streaming videollms for real-time procedural video understanding","venue":null,"work_id":"74bc44fa-7cf0-445e-a837-49c5d4faab49","year":2025},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-08-14T05:50:11.493950Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"cited_paper":"/paper/2504.13915","citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:a5e51b3b32ab992ee103a6adf22e29a2492d1486ee4f2f117a374cda05ffc10b","observation_id":"aae0a56c-695d-47e3-9cfc-69be4f57db2e","resolution":{"observed_at":"2026-05-11T11:31:03.596964Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.15595","last_updated":"2023-06-28T04:26:05Z","snapshot_observed_at":"2026-08-16T11:24:28.964729Z","submitted_at":"2023-06-27T16:26:26Z","title":"Extending Context Window of Large Language Models via Positional Interpolation","version":2},"cited_work":{"arxiv_id":"2306.15595","doi":"10.48550/arxiv.2306.15595","metadata_source":"pith","pith_arxiv_id":"2306.15595","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Extending Context Window of Large Language Models via Positional Interpolation","venue":"cs.CL","work_id":"c8b6df85-e7da-4bd8-90a4-d309cc2a0f60","year":2023},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-08-14T05:50:11.493950Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"cited_paper":"/paper/2306.15595","citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:1b47c0ebf813be836ef2fbca94ab8fad3591e44c0cad4c3679d76047a8b0e45c","observation_id":"bb0993fe-c244-4af7-ab40-485b7c61345d","resolution":{"observed_at":"2026-05-13T10:20:58.016961Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-15T03:08:14.561303+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-15T03:08:14.561303+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.12769","last_updated":"2025-03-17T03:05:31Z","snapshot_observed_at":"2026-08-16T12:49:30.665240Z","submitted_at":"2025-03-17T03:05:31Z","title":"ViSpeak: Visual Instruction Feedback in Streaming Videos","version":1},"cited_work":{"arxiv_id":"2503.12769","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.12769","snapshot_observed_at":"2026-07-02T19:07:17.481649Z","title":"Vispeak: Visual instruction feedback in streaming videos","venue":null,"work_id":"ff02f96e-6220-439c-a2d7-ebe7d4b79876","year":2025},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-08-14T05:50:11.493950Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"cited_paper":"/paper/2503.12769","citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:ee7497754cd1c59fd621741bb138128f5699d1c5ac2fddc8d9c3076988d893ce","observation_id":"c43ba94c-104b-4986-8759-33939e277c70","resolution":{"observed_at":"2026-05-11T11:31:03.582625Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Lm-infinite: Zero-shot extreme length generalization for large language models","venue":null,"work_id":"9a6d1e54-beca-4048-a619-ed17c8d66fe2","year":2024},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-08-14T05:50:11.493950Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:7b958b8f6581cccc45669db720478b16920058340499253986ca4aad248ac970","observation_id":"366c7205-f8de-4192-b100-b535d2072998","resolution":{"observed_at":"2026-05-18T11:46:19.677859Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2506.15745","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T16:39:58.347155Z","title":"Infinipot-v: Memory-constrained kv cache com- pression for streaming video understanding","venue":null,"work_id":"798f7dbf-55e6-44c0-9508-6aec9da8c639","year":2025},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-08-14T05:50:11.493950Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:97ea08aedfd61c6d4499118b3cee5558c9eea2a2aadb6e55d0fdf12a5de0bb2e","observation_id":"912e9849-5a85-46c4-ada0-1164ec692f17","resolution":{"observed_at":"2026-05-11T11:31:03.592378Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03326","last_updated":"2024-10-26T16:35:13Z","snapshot_observed_at":"2026-08-18T11:56:50.710310Z","submitted_at":"2024-08-06T17:59:44Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","version":3},"cited_work":{"arxiv_id":"2408.03326","doi":"10.48550/arxiv.2408.03326","metadata_source":"pith","pith_arxiv_id":"2408.03326","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","venue":"cs.CV","work_id":"f5f2452b-f2a9-49ac-b38d-c76e18cdfe49","year":2024},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-08-14T05:50:11.493950Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"cited_paper":"/paper/2408.03326","citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:5087253be7cdb2777301678144a0a1f04afae146695eed95b9832a5a7765d637","observation_id":"f52bea55-7459-4551-a43f-d5c408a2e669","resolution":{"observed_at":"2026-05-11T11:31:03.558420Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03628","last_updated":"2024-11-06T02:50:30Z","snapshot_observed_at":"2026-08-19T20:35:22.804334Z","submitted_at":"2024-11-06T02:50:30Z","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2411.03628","doi":"10.48550/arxiv.2411.03628","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.03628","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Streamingbench: Assessing the gap for mllms to achieve streaming video un- derstanding","venue":"arXiv (Cornell University)","work_id":"caec985b-dd2d-4dbb-9199-d147732de99a","year":2024},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-08-14T05:50:11.493950Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"cited_paper":"/paper/2411.03628","citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:14e02a552d40d7275e96f06aa5a940881011d0b1340195b0c98ce7680c581930","observation_id":"eb3684ab-300c-44e6-a7d0-7b730bf49737","resolution":{"observed_at":"2026-05-11T11:31:03.575498Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.03172","last_updated":"2023-11-20T23:09:34Z","snapshot_observed_at":"2026-07-06T15:51:12.179086Z","submitted_at":"2023-07-06T17:54:11Z","title":"Lost in the Middle: How Language Models Use Long Contexts","version":3},"cited_work":{"arxiv_id":"2307.03172","doi":"10.1162/tacl","metadata_source":"pith","pith_arxiv_id":"2307.03172","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Lost in the Middle: How Language Models Use Long Contexts","venue":"cs.CL","work_id":"37c05e13-4a24-44f8-a1c4-da1bbe7223aa","year":2023},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-08-14T05:50:11.493950Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"cited_paper":"/paper/2307.03172","citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:0fe749a2cc942513de9e1760ee793d70ac7eae6273729ec0474bbd10af5c3466","observation_id":"f391ef84-cbb1-40bb-bf2f-1326a2d7e83a","resolution":{"observed_at":"2026-05-11T11:31:03.570719Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15269","last_updated":"2026-04-23T12:54:38Z","snapshot_observed_at":"2026-08-12T15:58:40.311449Z","submitted_at":"2025-05-21T08:47:15Z","title":"LiveVLM: Efficient Online Video Understanding via Streaming-Oriented KV Cache and Retrieval","version":2},"cited_work":{"arxiv_id":"2505.15269","doi":"10.48550/arxiv.2505.15269","metadata_source":"pith","pith_arxiv_id":"2505.15269","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"LiveVLM: Efficient Online Video Understanding via Streaming-Oriented KV Cache and Retrieval","venue":"cs.CV","work_id":"a00b4d7a-5adc-4855-89a1-5356dc946a85","year":2025},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-08-14T05:50:11.493950Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"cited_paper":"/paper/2505.15269","citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:e41dece12cc4203eca9c067235c1a235a3e33d0f0656c57c6d1d5301f5c0d160","observation_id":"813374e9-2614-4a74-b24e-6d7c58b750b5","resolution":{"observed_at":"2026-05-11T11:31:03.587336Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.02546","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"On discrim- inative vs","venue":null,"work_id":"e6049f41-8edb-46d7-a337-6b57a271c937","year":null},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-08-14T05:50:11.493950Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:7ddbd32345a3710422c8baf0e4eae46ce3dde6030a0cac7e7a6cdfeef6699efb","observation_id":"6ce05c96-c77c-4e1f-be56-354d722a8297","resolution":{"observed_at":"2026-05-11T11:31:03.612285Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2108.12409","last_updated":"2022-04-22T18:20:48Z","snapshot_observed_at":"2026-08-07T11:26:24.970964Z","submitted_at":"2021-08-27T17:35:06Z","title":"Train Short, Test Long: Attention with Linear Biases Enables Input Length Extrapolation","version":2},"cited_work":{"arxiv_id":"2108.12409","doi":"10.48550/arxiv.2108.12409","metadata_source":"pith","pith_arxiv_id":"2108.12409","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Train Short, Test Long: Attention with Linear Biases Enables Input Length Extrapolation","venue":"cs.CL","work_id":"145b1374-5258-4c00-a433-4db0f5a50749","year":2021},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-08-14T05:50:11.493950Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"cited_paper":"/paper/2108.12409","citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:de975e03e78b936248c8398d91b66cde33e5a537cd689ac6e2c45ac3893b3496","observation_id":"c3db2087-ef04-4486-b0e0-84246ad47a2e","resolution":{"observed_at":"2026-05-13T01:01:23.440331Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05530","last_updated":"2024-12-16T17:39:39Z","snapshot_observed_at":"2026-08-14T18:15:53.516440Z","submitted_at":"2024-03-08T18:54:20Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","version":5},"cited_work":{"arxiv_id":"2403.05530","doi":"10.48550/arxiv.2403.05530","metadata_source":"pith","pith_arxiv_id":"2403.05530","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","venue":"cs.CL","work_id":"80e3e977-f1bb-4c83-8d0c-1ab0a0c5c3f1","year":2024},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-08-14T05:50:11.493950Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"cited_paper":"/paper/2403.05530","citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:f6f84ed7e3d4c09da0b1e31fba72e02dffa9cb3b964a49d93fdc1e78defcad3a","observation_id":"9e0f6efa-4005-45a8-85ac-57d3c18dbf16","resolution":{"observed_at":"2026-05-11T11:31:03.649723Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-02T03:08:14.426583+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T03:08:14.426583+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.05467","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-02T17:27:15.523830Z","title":"Streambridge: Turning your offline video large language model into a proactive streaming assistant","venue":null,"work_id":"6f0d7f4a-5862-4b5f-8e70-eeaff1ed8a54","year":2025},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-08-14T05:50:11.493950Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:9eb5a233977da92459556426c28da236ac8e829587dec344ac6fb2f94efffc60","observation_id":"ca99b3bd-dc4e-42e5-86f6-5928c1a9bfbf","resolution":{"observed_at":"2026-05-11T11:31:03.617054Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.17453","last_updated":"2024-04-07T00:56:53Z","snapshot_observed_at":"2026-08-14T06:01:54.549199Z","submitted_at":"2023-09-29T17:59:56Z","title":"Efficient Streaming Language Models with Attention Sinks","version":4},"cited_work":{"arxiv_id":"2309.17453","doi":"10.48550/arxiv.2309.17453","metadata_source":"pith","pith_arxiv_id":"2309.17453","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Efficient Streaming Language Models with Attention Sinks","venue":"cs.CL","work_id":"a8d25452-c237-48c9-88a4-682717c3979a","year":2023},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-08-14T05:50:11.493950Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"cited_paper":"/paper/2309.17453","citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:df47118f9ca276ad6731b6bad8d42d01e0969b6e68043cb30191741136e21aac","observation_id":"b94c4bb3-fada-4c4a-909f-4b6b4acbe0cd","resolution":{"observed_at":"2026-05-11T11:31:03.621621Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.09608","last_updated":"2025-10-10T17:59:58Z","snapshot_observed_at":"2026-08-19T16:53:15.380005Z","submitted_at":"2025-10-10T17:59:58Z","title":"StreamingVLM: Real-Time Understanding for Infinite Video Streams","version":1},"cited_work":{"arxiv_id":"2510.09608","doi":"10.48550/arxiv.2510.09608","metadata_source":"pith","pith_arxiv_id":"2510.09608","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"StreamingVLM: Real-Time Understanding for Infinite Video Streams","venue":"cs.CV","work_id":"e6785f4f-90d8-45ab-8e34-1a406ea2db88","year":2025},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-08-14T05:50:11.493950Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"cited_paper":"/paper/2510.09608","citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:ba5a65e8e9097316965f926ad7a878ec12bf5e36655b2912c8e8b2cbf41c77c8","observation_id":"63a5bf8b-564b-42e9-a4b0-665af66bd598","resolution":{"observed_at":"2026-05-17T11:51:33.451476Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.15717","last_updated":"2025-08-21T16:56:29Z","snapshot_observed_at":"2026-08-18T16:22:02.091795Z","submitted_at":"2025-08-21T16:56:29Z","title":"StreamMem: Query-Agnostic KV Cache Memory for Streaming Video Understanding","version":1},"cited_work":{"arxiv_id":"2508.15717","doi":"10.48550/arxiv.2508.15717","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.15717","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Streammem: Query-agnostic kv cache memory for stream- ing video understanding.arXiv preprint arXiv:2508.15717","venue":"ArXiv.org","work_id":"752bf839-d545-4378-a68d-4a958ba10cee","year":2025},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-08-14T05:50:11.493950Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"cited_paper":"/paper/2508.15717","citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:1ef9a5ea3c57ab1313e9b3e9b9876dcd2d4998dacd0333613368d99bc74f0900","observation_id":"f4b025e5-d85a-49d5-9bda-3efef210333c","resolution":{"observed_at":"2026-05-11T11:31:03.635026Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08085","last_updated":"2024-06-30T05:39:46Z","snapshot_observed_at":"2026-08-16T13:43:48.910620Z","submitted_at":"2024-06-12T11:07:55Z","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","version":2},"cited_work":{"arxiv_id":"2406.08085","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.08085","snapshot_observed_at":"2026-07-04T20:00:08.182505Z","title":"Flash-vstream: Memory-based real-time understanding for long video streams","venue":null,"work_id":"f53e5fea-3444-480b-aa38-be98378c0ced","year":2024},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-08-14T05:50:11.493950Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"cited_paper":"/paper/2406.08085","citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:213d6f36654ebf2d0f69728356cea6c19cdc6c2463ebfb62653da82154d61407","observation_id":"09c9c0bb-1fff-461f-aa8e-83141dc06e75","resolution":{"observed_at":"2026-05-11T11:31:03.562532Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"5bd94f06-d972-4ac0-a4fa-5fb96d69b9d9","year":2025},"citing_paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","snapshot_observed_at":"2026-08-14T05:50:11.493950Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-10T14:47:54.917408Z"},"links":{"citing_paper":"/paper/2605.01858"},"observation_digest":"sha256:b28f694228519578d4084ac430ddb01f67bbf0e11a8c77063e1542778e4cec1b","observation_id":"df74d09c-8af0-4303-a424-24aa76a10d28","resolution":{"observed_at":"2026-05-18T11:46:19.673538Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2605.01858","last_updated":"2026-05-03T13:02:44Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-14T05:50:11.493950Z","submitted_at":"2026-05-03T13:02:44Z","title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding"},"reference_resolution":{"displayed":19,"state_counts":{"malformed_identifier":0,"metadata_mismatch":5,"parse_uncertain":0,"unresolved":1,"verified_exact":12,"verified_fuzzy":1},"total_outbound_references":19},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"thesis":"As of 19 August 2026, this Paper Citation Record lists 19 of 19 outbound references and 1 inbound Pith citation observation for arXiv:2605.01858."}