{"as_of":"2026-08-09T05:51:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e1cad3b11f13022c9a98baa517d38d32b18a0e136df8a3384317af0c74dd4589","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":40,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":40,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":40,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":40,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T15:31:16.940927Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T00:29:15.346488Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2406.08035","last_updated":"2025-08-09T10:54:59Z","snapshot_observed_at":"2026-08-08T19:38:26.415599Z","submitted_at":"2024-06-12T09:36:52Z","title":"LVBench: An Extreme Long Video Understanding Benchmark","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-19T11:55:30.048525Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2406.08035"},"observation_digest":"sha256:9bb977b9b63c701bf47d1f32c37b600c73004f842f040b12b859f92cc8c2fe27","observation_id":"6afa325e-c4aa-4373-887a-6e3feca6923d","resolution":{"observed_at":"2026-05-19T11:55:30.200813Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2501.00574","last_updated":"2025-07-13T16:21:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-31T18:01:23Z","title":"VideoChat-Flash: Hierarchical Compression for Long-Context Video Modeling","version":4},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-18T04:02:43.261543Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2501.00574"},"observation_digest":"sha256:9ca646921b138b1c5ca7feb5b31411d34308fbcbe7dd8078581a3138339043de","observation_id":"fc2efdfa-89c7-4199-81f7-5ccd0e43f5b6","resolution":{"observed_at":"2026-05-18T04:02:43.448546Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2501.02955","last_updated":"2026-05-12T15:02:48Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-06T11:57:38Z","title":"MotionBench: Benchmarking and Improving Fine-grained Video Motion Understanding for Vision Language Models","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-23T05:44:31.546843Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2501.02955"},"observation_digest":"sha256:5523c9ffd2f81c3408e4e92437db2ba3345920b476c0d8fbf214755517a1ac2f","observation_id":"4c3f7916-40be-4917-86d3-699740523c97","resolution":{"observed_at":"2026-05-23T05:45:28.355335Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2501.13106","last_updated":"2025-06-03T03:33:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T18:59:46Z","title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","version":4},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-11T01:19:59.603343Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2501.13106"},"observation_digest":"sha256:33dde1c5b17087e2d2265bdc1e72bc134fea5670cf5afa0aa51cb080b89b15e7","observation_id":"a75c095b-8f2e-4bbb-815c-15164036c109","resolution":{"observed_at":"2026-05-11T01:20:00.287485Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-08-08T15:31:16.940927Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.06428","last_updated":"2025-02-11T14:59:25Z","snapshot_observed_at":"2026-08-08T15:25:00.286848Z","submitted_at":"2025-02-10T13:03:05Z","title":"CoS: Chain-of-Shot Prompting for Long Video Understanding","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-08T15:31:16.940927Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2502.06428"},"observation_digest":"sha256:e909057cc80f812d122a63e07478a0800067d4703f3b732e0ac2ae845a05e014","observation_id":"3ab614dd-274e-44cb-9262-13cb022a3a00","resolution":{"observed_at":"2026-08-08T15:31:16.940927Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2503.21776","last_updated":"2025-10-22T16:42:24Z","snapshot_observed_at":"2026-08-05T07:15:29.998948Z","submitted_at":"2025-03-27T17:59:51Z","title":"Video-R1: Reinforcing Video Reasoning in MLLMs","version":4},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-12T09:43:00.208065Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2503.21776"},"observation_digest":"sha256:2331f88c428b14bd9f51de2b1d12392cc0fbad7b9f90897701c98a0d43cbc569","observation_id":"7b31e901-5b45-4c98-bc84-9819d5b33790","resolution":{"observed_at":"2026-05-12T09:43:00.408504Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2504.05299","last_updated":"2025-04-07T17:58:57Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-07T17:58:57Z","title":"SmolVLM: Redefining small and efficient multimodal models","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-13T20:23:50.552549Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2504.05299"},"observation_digest":"sha256:373c3a2c2a9f6454571d713bdeb0d01d3d86149b3e654ca79133755cddcdece8","observation_id":"3cd6b174-5f79-4bb2-b775-89284f17b9df","resolution":{"observed_at":"2026-05-13T20:23:51.734057Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-08-07T15:20:57.694251Z","title":"Kangaroo: A powerful video-language model supporting long-context video input.arXiv preprint arXiv:2408.15542, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15447","last_updated":"2025-05-21T12:29:40Z","snapshot_observed_at":"2026-08-09T00:23:31.907645Z","submitted_at":"2025-05-21T12:29:40Z","title":"ViaRL: Adaptive Temporal Grounding via Visual Iterated Amplification Reinforcement Learning","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T15:20:57.694251Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2505.15447"},"observation_digest":"sha256:6e62521ebcc8f2872b30b697d758e50d348350eec660ced53fbfb5b694120e62","observation_id":"8335557e-cb75-467b-b018-6d8712a4a5b0","resolution":{"observed_at":"2026-08-07T15:20:57.694251Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-08-07T15:20:45.733144Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15529","last_updated":"2025-05-21T13:52:17Z","snapshot_observed_at":"2026-08-07T15:13:30.827650Z","submitted_at":"2025-05-21T13:52:17Z","title":"Clapper: Compact Learning and Video Representation in VLMs","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-07T15:20:45.733144Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2505.15529"},"observation_digest":"sha256:5ea5dd29b0021c294aa7e7984539ce7ac690be4eef516fc4d3617a550824d87e","observation_id":"81ccf352-fc30-40ad-a176-ed4d766704f4","resolution":{"observed_at":"2026-08-07T15:20:45.733144Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-08-07T14:03:01.765688Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20124","last_updated":"2025-05-27T12:10:27Z","snapshot_observed_at":"2026-08-08T09:10:32.146078Z","submitted_at":"2025-05-26T15:24:06Z","title":"TUNA: Comprehensive Fine-grained Temporal Understanding Evaluation on Dense Dynamic Videos","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-07T14:03:01.765688Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2505.20124"},"observation_digest":"sha256:bde36e2ecc7191a5d21d9a2a0fe6c1a5cf50dfc3f6d38d2b4324363893dbe09c","observation_id":"06c314bb-4e5b-4358-8610-210596f6fd2f","resolution":{"observed_at":"2026-08-07T14:03:01.765688Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-08-07T14:02:55.535106Z","title":"Kangaroo: A powerful video-language model supporting long-context video input.arXiv preprint arXiv:2408.15542, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20256","last_updated":"2025-05-26T17:34:06Z","snapshot_observed_at":"2026-08-07T13:53:51.495895Z","submitted_at":"2025-05-26T17:34:06Z","title":"Omni-R1: Reinforcement Learning for Omnimodal Reasoning via Two-System Collaboration","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T14:02:55.535106Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2505.20256"},"observation_digest":"sha256:d88eefb89f235f61b062df09b8979ce096fa88c904936b4ccf764d1fb3a5baf7","observation_id":"a3fcd63b-ceaa-46f9-bd7f-2fcd422481f8","resolution":{"observed_at":"2026-08-07T14:02:55.535106Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-08-07T12:22:13.508706Z","title":"Kangaroo: A powerful video-language model supporting long-context video input,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24718","last_updated":"2025-06-08T14:43:47Z","snapshot_observed_at":"2026-08-09T05:12:16.591061Z","submitted_at":"2025-05-30T15:42:19Z","title":"Reinforcing Video Reasoning with Focused Thinking","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T12:22:13.508706Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2505.24718"},"observation_digest":"sha256:888652dc1893f9d5bdbc3f1c00577eaf14bb475416d651e615bd3dabe6f0eabc","observation_id":"6443ad05-adcd-41d2-b2e6-1dc1fa460b2e","resolution":{"observed_at":"2026-08-07T12:22:13.508706Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-08-07T11:59:07.645844Z","title":"Kangaroo: A powerful video-language model supporting long-context video input, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.00993","last_updated":"2025-06-01T12:49:39Z","snapshot_observed_at":"2026-08-08T18:44:33.509277Z","submitted_at":"2025-06-01T12:49:39Z","title":"FlexSelect: Flexible Token Selection for Efficient Long Video Understanding","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T11:59:07.645844Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2506.00993"},"observation_digest":"sha256:fb7d9a8c3663fa3e7b0ecfd5fa71f85f75b495a0fd312ed7246692c627ffe314","observation_id":"45e15b40-6cc5-4825-8b1a-1dea06352fcd","resolution":{"observed_at":"2026-08-07T11:59:07.645844Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-08-07T11:51:27.753715Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01274","last_updated":"2026-06-11T17:06:50Z","snapshot_observed_at":"2026-08-08T07:40:04.746667Z","submitted_at":"2025-06-02T03:08:07Z","title":"ReFoCUS: Reinforcement-guided Frame Optimization for Contextual Understanding","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T11:51:27.753715Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2506.01274"},"observation_digest":"sha256:cb042761a86464d8b7f2b344ab7bb17ae7a8bbeee6d921d0422ef5b2d9570376","observation_id":"de8b4712-1d7a-4f71-9dcf-3612323dec24","resolution":{"observed_at":"2026-08-07T11:51:27.753715Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-08-07T11:52:03.120334Z","title":"Kangaroo: A powerful video-language model supporting long-context video input.arXiv preprint arXiv:2408.15542, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01300","last_updated":"2025-06-02T04:23:21Z","snapshot_observed_at":"2026-08-07T11:42:36.031718Z","submitted_at":"2025-06-02T04:23:21Z","title":"ReAgent-V: A Reward-Driven Multi-Agent Framework for Video Understanding","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:03.120334Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2506.01300"},"observation_digest":"sha256:212740a55d53fd7ebb691120a57dc40c69d3eddce481080f56e24d99a629549b","observation_id":"82c0dd4b-7efa-417e-8b24-09987e4f0f69","resolution":{"observed_at":"2026-08-07T11:52:03.120334Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2506.01844","last_updated":"2025-06-02T16:30:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-02T16:30:19Z","title":"SmolVLA: A Vision-Language-Action Model for Affordable and Efficient Robotics","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-11T21:22:36.902119Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2506.01844"},"observation_digest":"sha256:ed13a0673bf634d0908a402f5c61b88ac7d176f7ca6c1d62240acbc8b5cfa8b1","observation_id":"c38d776f-7f12-467b-b91e-4bc3b14c4a9a","resolution":{"observed_at":"2026-05-11T21:22:37.480720Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-08-07T11:35:46.174675Z","title":"Kangaroo: A powerful video-language model supporting long-context video input.arXiv preprint arXiv:2408.15542, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01908","last_updated":"2025-06-02T17:28:26Z","snapshot_observed_at":"2026-08-07T11:29:22.385642Z","submitted_at":"2025-06-02T17:28:26Z","title":"Reinforcement Learning Tuning for VideoLLMs: Reward Design and Data Efficiency","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T11:35:46.174675Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2506.01908"},"observation_digest":"sha256:b135e5e709fef7aba66033c15c0f55252106d1abc47e6cd372daccc041d846de","observation_id":"c7ba70d3-efd9-41c4-bb52-345f72de1a67","resolution":{"observed_at":"2026-08-07T11:35:46.174675Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-08-06T23:48:37.373080Z","title":"Kangaroo: A powerful video-language model supporting long-context video input.arXiv preprint arXiv:2408.15542, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.16141","last_updated":"2025-06-19T08:49:13Z","snapshot_observed_at":"2026-08-08T06:18:41.273900Z","submitted_at":"2025-06-19T08:49:13Z","title":"GRPO-CARE: Consistency-Aware Reinforcement Learning for Multimodal Reasoning","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T23:48:37.373080Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2506.16141"},"observation_digest":"sha256:f71c2bd52633e48fc90b3ad8945e650a44259668f64c0d52faec8c67bcf9bdfb","observation_id":"d2d036f9-03e3-4183-a4e4-e3dade6832b8","resolution":{"observed_at":"2026-08-06T23:48:37.373080Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-08-06T22:15:03.351183Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.22139","last_updated":"2025-07-22T07:42:31Z","snapshot_observed_at":"2026-08-06T22:07:43.493361Z","submitted_at":"2025-06-27T11:30:51Z","title":"Q-Frame: Query-aware Frame Selection and Multi-Resolution Adaptation for Video-LLMs","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T22:15:03.351183Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2506.22139"},"observation_digest":"sha256:4a437cf645a20083cb68b3c4243890da0f4409b546a1972aad01d0b23f9a5b2a","observation_id":"0f33d806-78f9-4ccf-9bd3-34caa259491c","resolution":{"observed_at":"2026-08-06T22:15:03.351183Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-08-06T21:37:05.338028Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.23825","last_updated":"2025-07-24T07:25:10Z","snapshot_observed_at":"2026-08-08T09:10:29.941912Z","submitted_at":"2025-06-30T13:17:49Z","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:05.338028Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2506.23825"},"observation_digest":"sha256:42a3e1e7bd47aef9c96da20f1becd53af9816aacc51dbc99bb2acbe051703e18","observation_id":"0f4f6773-ee34-41bd-bd60-e24713ee2a90","resolution":{"observed_at":"2026-08-06T21:37:05.338028Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-08-06T14:05:32.407540Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19821","last_updated":"2025-08-02T13:22:34Z","snapshot_observed_at":"2026-08-08T10:33:30.479422Z","submitted_at":"2025-07-26T06:38:07Z","title":"LAVA: Language Driven Scalable and Versatile Traffic Video Analytics","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T14:05:32.407540Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2507.19821"},"observation_digest":"sha256:0a9e6ae52143ea3db9b7707b34b43c13e2a0e1afcc40b406ceb6b9fdf6545e85","observation_id":"79cdbcc8-6c07-48b9-9e72-e300b01626e0","resolution":{"observed_at":"2026-08-06T14:05:32.407540Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-08-05T15:40:40.438516Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.19650","last_updated":"2025-08-29T02:25:23Z","snapshot_observed_at":"2026-08-07T05:11:55.108109Z","submitted_at":"2025-08-27T07:58:16Z","title":"Video-LevelGauge: Investigating Contextual Positional Bias in Large Video Language Models","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-05T15:40:40.438516Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2508.19650"},"observation_digest":"sha256:e42adbe1fc4088c46b99b6ebb21523f8e016666de2f9c202d0f0ec2e6d69f189","observation_id":"adce1126-a8a0-4236-b4ca-617e4f826596","resolution":{"observed_at":"2026-08-05T15:40:40.438516Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-08-05T15:10:16.769169Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.20478","last_updated":"2026-05-29T09:48:44Z","snapshot_observed_at":"2026-08-05T15:10:01.360889Z","submitted_at":"2025-08-28T06:55:08Z","title":"Video-MTR: Reinforced Multi-Turn Reasoning for Long Video Understanding","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-05T15:10:16.769169Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2508.20478"},"observation_digest":"sha256:23504f24180115a5f164684d26b6bdd2c651cc950b78b8f0a4e5da413ca55b01","observation_id":"83510f07-e3d2-4591-9f11-3089933d292e","resolution":{"observed_at":"2026-08-05T15:10:16.769169Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-08-04T21:27:34.988625Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.07680","last_updated":"2025-09-09T17:59:39Z","snapshot_observed_at":"2026-08-04T21:27:30.632810Z","submitted_at":"2025-09-09T17:59:39Z","title":"CAViAR: Critic-Augmented Video Agentic Reasoning","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-04T21:27:34.988625Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2509.07680"},"observation_digest":"sha256:527b069523c021bba0d42b7ca00d8b6ad3e81e444104ea4ff8e73ffc30cd284c","observation_id":"9faf2417-4896-46ee-9883-69e0bdcc3e56","resolution":{"observed_at":"2026-08-04T21:27:34.988625Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2512.01707","last_updated":"2026-05-13T04:12:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T14:15:44Z","title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-17T02:49:12.987772Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2512.01707"},"observation_digest":"sha256:9dfc4e86d1b6ea660e4f24e74d01654b631e691c9f06b43eedf99e705aba5fe2","observation_id":"b5c68ae3-7955-4fa6-b102-f464b8aa660a","resolution":{"observed_at":"2026-05-17T02:51:28.132505Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2601.14724","last_updated":"2026-05-07T12:10:26Z","snapshot_observed_at":"2026-08-05T03:02:47.238051Z","submitted_at":"2026-01-21T07:26:15Z","title":"HERMES: KV Cache as Hierarchical Memory for Efficient Streaming Video Understanding","version":4},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-16T12:55:04.564442Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2601.14724"},"observation_digest":"sha256:3e9b4c1b3cf13349490f113dbd00307bf129d7bc39119b0d47e0c517cdf738bf","observation_id":"8313bc12-2f58-443d-87b3-565b183979ff","resolution":{"observed_at":"2026-05-16T12:57:53.873910Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2604.08120","last_updated":"2026-04-09T11:40:25Z","snapshot_observed_at":"2026-07-06T22:57:18.202215Z","submitted_at":"2026-04-09T11:40:25Z","title":"Small Vision-Language Models are Smart Compressors for Long Video Understanding","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-10T18:38:49.654073Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2604.08120"},"observation_digest":"sha256:2c70ba809611b4982d18b0e9178527074292ef38cc10a7de9ec81b868966938a","observation_id":"d1c7aabf-cc5c-4e9e-adc7-4a15863bba87","resolution":{"observed_at":"2026-05-11T00:15:51.479344Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2604.14149","last_updated":"2026-04-16T15:48:38Z","snapshot_observed_at":"2026-07-06T23:02:00.082783Z","submitted_at":"2026-04-15T17:59:52Z","title":"One Token per Highly Selective Frame: Towards Extreme Compression for Long Video Understanding","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-10T13:28:58.920442Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2604.14149"},"observation_digest":"sha256:623f0b5e3852e46d33a7ce3b6e224dcf8a5fe60915dbb529f07d016f7de392cc","observation_id":"82579acb-6fab-499a-978b-dcefacfa5f2e","resolution":{"observed_at":"2026-05-10T13:30:26.569125Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2604.17052","last_updated":"2026-04-18T16:22:05Z","snapshot_observed_at":"2026-08-08T21:49:06.109128Z","submitted_at":"2026-04-18T16:22:05Z","title":"OASIS: On-Demand Hierarchical Event Memory for Streaming Video Reasoning","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-10T06:51:52.861981Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2604.17052"},"observation_digest":"sha256:ced402b8dae5bbb0da94fd548e8d3392e8af08c0724cb4ca06c5e504eb23d5e1","observation_id":"3e04ced6-a4e9-4ed8-b2f9-66302f455250","resolution":{"observed_at":"2026-05-10T06:56:47.759603Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2605.00444","last_updated":"2026-05-01T06:24:40Z","snapshot_observed_at":"2026-07-06T23:13:52.304925Z","submitted_at":"2026-05-01T06:24:40Z","title":"Scaling Video Understanding via Compact Latent Multi-Agent Collaboration","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-09T20:11:11.410051Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2605.00444"},"observation_digest":"sha256:b3d018be659ae5c283e07b838efcc42bbce4cf1eada6450ecc3db30689c58118","observation_id":"92803c71-da0a-4e87-a538-1536a6e706fd","resolution":{"observed_at":"2026-05-11T15:21:09.522561Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2605.07575","last_updated":"2026-05-11T11:58:53Z","snapshot_observed_at":"2026-07-06T23:19:56.415523Z","submitted_at":"2026-05-08T10:46:10Z","title":"Response-G1: Explicit Scene Graph Modeling for Proactive Streaming Video Understanding","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-05-11T02:30:55.939351Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2605.07575"},"observation_digest":"sha256:d67e68f67a58d10bd6e9cf118bd9c672f369eb2b9fd7c2da7661d70fcda02a4e","observation_id":"c8418107-8250-4579-88e0-92a470464e44","resolution":{"observed_at":"2026-05-11T03:20:56.479634Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2605.07575","last_updated":"2026-05-11T11:58:53Z","snapshot_observed_at":"2026-07-06T23:19:56.415523Z","submitted_at":"2026-05-08T10:46:10Z","title":"Response-G1: Explicit Scene Graph Modeling for Proactive Streaming Video Understanding","version":2},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-05-12T03:00:34.728880Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2605.07575"},"observation_digest":"sha256:e4c1af4b6ac5985a7b25b1c563523692c57fd63197b6baa4c158284140148fa4","observation_id":"c1005e51-57e9-4c04-9b1d-862c7d6bfbde","resolution":{"observed_at":"2026-05-12T03:01:17.700487Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2605.17921","last_updated":"2026-06-01T06:29:58Z","snapshot_observed_at":"2026-07-06T23:28:50.117638Z","submitted_at":"2026-05-18T06:29:44Z","title":"An Efficient Streaming Video Understanding Framework with Agentic Control","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-05-20T11:30:22.151045Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2605.17921"},"observation_digest":"sha256:ef1b658dd261f3c2095cb7208474bdd2d75501b71041c4354a3f8b85e72a3dc5","observation_id":"88c56c0f-3dab-484a-878e-e22f4a3402a5","resolution":{"observed_at":"2026-05-20T11:33:14.452976Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2605.25621","last_updated":"2026-05-25T09:23:19Z","snapshot_observed_at":"2026-07-06T23:35:36.158984Z","submitted_at":"2026-05-25T09:23:19Z","title":"StreamOV: Streaming Omni-Video Understanding via Evidence-Guided Memory and Response Triggering","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-06-29T22:27:17.092553Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2605.25621"},"observation_digest":"sha256:33b549b03f7e39321e4bd8bcf94513867eb4dadf6524be11db7071048c8a76ef","observation_id":"06199d9c-9c90-4ba0-bca6-6dfab8cdebb0","resolution":{"observed_at":"2026-06-29T22:34:02.071338Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2606.12195","last_updated":"2026-06-10T15:17:08Z","snapshot_observed_at":"2026-08-01T02:09:41.655807Z","submitted_at":"2026-06-10T15:17:08Z","title":"InternVideo3: Agentify Foundation Models with Multimodal Contextual Reasoning","version":1},"reference_index":296,"source":"arxiv_source","source_observed_at":"2026-06-27T09:48:27.652901Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2606.12195"},"observation_digest":"sha256:6fab2e092af3f4dc37b4f7936f32adad28e9fc173aa3959eae2627c9e18470f8","observation_id":"4f5b8170-f9d8-43a5-82cc-fa3598a11b04","resolution":{"observed_at":"2026-07-03T10:48:03.192666Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2606.18441","last_updated":"2026-06-16T19:42:54Z","snapshot_observed_at":"2026-08-03T17:38:18.858871Z","submitted_at":"2026-06-16T19:42:54Z","title":"Reasoning as Intersection: Consensus-Frame Alignment for Visual Focus in Video-MLLMs","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-06-27T01:02:37.632461Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2606.18441"},"observation_digest":"sha256:9233d23d06b919661d4a9985d4133212a4a8dba0e9e718f61049c479994342e1","observation_id":"ed14d6ca-5e62-4cb0-9eed-d7f32f883594","resolution":{"observed_at":"2026-07-03T20:58:57.726244Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":"2408.15542","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-04T00:29:15.346488Z","title":"Kangaroo: A powerful video-language model supporting long-context video input","venue":null,"work_id":"5e709eaf-2719-4a1e-8ee6-1d5f6f1ac2fd","year":2024},"citing_paper":{"arxiv_id":"2606.19341","last_updated":"2026-07-17T18:52:18Z","snapshot_observed_at":"2026-08-06T12:20:24.096310Z","submitted_at":"2026-06-17T17:59:56Z","title":"Native Active Perception as Reasoning for Omni-Modal Understanding","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-26T21:14:11.297384Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2606.19341"},"observation_digest":"sha256:c8be1c08e8b936795428e42bcf5a8a4ee5ce26d7b69a92962cb7a9a1d28096b5","observation_id":"a605ff3f-0bea-437f-a1a5-671a2adbaa57","resolution":{"observed_at":"2026-07-04T00:29:15.349201Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-08-02T10:57:19.810078Z","title":"Video-LLaV A: Learning united visual representation by alignment before projection","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.19341","last_updated":"2026-07-17T18:52:18Z","snapshot_observed_at":"2026-08-06T12:20:24.096310Z","submitted_at":"2026-06-17T17:59:56Z","title":"Native Active Perception as Reasoning for Omni-Modal Understanding","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-02T10:57:19.810078Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2606.19341"},"observation_digest":"sha256:caf2fe7b29bfd2fecc6b6efb628bb3b33ae2fee173e988d5f6b8023cbe9594d0","observation_id":"71114075-e508-49b7-8ba5-8ebae0751e93","resolution":{"observed_at":"2026-08-02T10:57:19.810078Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-07-12T09:09:15.248815Z","title":"Kangaroo: A powerful video-language model supporting long-context video input.arXiv preprint arXiv:2408.15542, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.02607","last_updated":"2026-07-01T13:51:02Z","snapshot_observed_at":"2026-07-12T09:09:14.587474Z","submitted_at":"2026-07-01T13:51:02Z","title":"Latent Visual Cache for Video Reasoning","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-07-12T09:09:15.248815Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2607.02607"},"observation_digest":"sha256:9a43bc18773e4a0e4f6e49728c90f0da15e5ba318a096860b4ae99e1fba51d59","observation_id":"01d453a6-8134-402f-ba94-496cc1c508df","resolution":{"observed_at":"2026-07-12T09:09:15.248815Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15542","snapshot_observed_at":"2026-08-01T08:49:13.271024Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.21013","last_updated":"2026-07-25T07:42:56Z","snapshot_observed_at":"2026-08-07T08:24:32.757054Z","submitted_at":"2026-07-23T07:52:40Z","title":"EmoAgent-R1: Towards Multimodal Emotion Understanding with Reinforcement Learning-based Dynamic Agent Specialization","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-01T08:49:13.271024Z"},"links":{"cited_paper":"/paper/2408.15542","citing_paper":"/paper/2607.21013"},"observation_digest":"sha256:b5e1eca2fea7791c53273da47cabb2d537060ef9daf195d00a5faa42d7bc304a","observation_id":"30cccac3-b3ac-40f9-adef-3c640cb6616b","resolution":{"observed_at":"2026-08-01T08:49:13.271024Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2408.15542/citation-record","integrity":"/paper/2408.15542/integrity","json":"/paper/2408.15542/citation-record.json","paper":"/paper/2408.15542"},"outbound":[],"paper":{"arxiv_id":"2408.15542","last_updated":"2024-08-28T05:34:14Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T19:06:54.428910Z","submitted_at":"2024-08-28T05:34:14Z","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 40 inbound Pith citation observations for arXiv:2408.15542."}