{"as_of":"2026-08-07T21:24:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:f4452b4228e26f22dff1ab51e42e526e7fa52aca9fa661baab210fce74df9923","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":27,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":27,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":27,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":27,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:29:59.428434Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T06:39:37.386392Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2404.03413","last_updated":"2024-04-04T12:46:01Z","snapshot_observed_at":"2026-07-06T17:55:36.748672Z","submitted_at":"2024-04-04T12:46:01Z","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","version":1},"cited_work":{"arxiv_id":"2404.03413","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.03413","snapshot_observed_at":"2026-07-04T06:39:37.386392Z","title":"Minigpt4-video: Advancing multimodal llms for video understanding with interleaved visual-textual tokens","venue":null,"work_id":"cc937528-86d1-430f-bb5d-4980dbaadd72","year":2024},"citing_paper":{"arxiv_id":"2406.04264","last_updated":"2025-01-01T15:53:58Z","snapshot_observed_at":"2026-08-03T20:38:36.602554Z","submitted_at":"2024-06-06T17:09:32Z","title":"MLVU: Benchmarking Multi-task Long Video Understanding","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-14T19:55:26.333923Z"},"links":{"cited_paper":"/paper/2404.03413","citing_paper":"/paper/2406.04264"},"observation_digest":"sha256:414e491339926341f06fdc389dc4bd05805acb705616e7ff3bf2274650af703b","observation_id":"ccdaa848-0a74-47df-a5d9-781b71649ec2","resolution":{"observed_at":"2026-05-14T19:55:26.511432Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03413","last_updated":"2024-04-04T12:46:01Z","snapshot_observed_at":"2026-07-06T17:55:36.748672Z","submitted_at":"2024-04-04T12:46:01Z","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","version":1},"cited_work":{"arxiv_id":"2404.03413","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.03413","snapshot_observed_at":"2026-07-04T06:39:37.386392Z","title":"Minigpt4-video: Advancing multimodal llms for video understanding with interleaved visual-textual tokens","venue":null,"work_id":"cc937528-86d1-430f-bb5d-4980dbaadd72","year":2024},"citing_paper":{"arxiv_id":"2406.07476","last_updated":"2024-10-30T06:49:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-11T17:22:23Z","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-11T02:44:53.284345Z"},"links":{"cited_paper":"/paper/2404.03413","citing_paper":"/paper/2406.07476"},"observation_digest":"sha256:c328bcb35de3c37ceaa1bf269a6a1689006b39f93192b22bf3521d8833e49ae7","observation_id":"1c5b397b-649d-401b-a0f3-626d2b80f660","resolution":{"observed_at":"2026-05-11T02:44:53.352805Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03413","last_updated":"2024-04-04T12:46:01Z","snapshot_observed_at":"2026-07-06T17:55:36.748672Z","submitted_at":"2024-04-04T12:46:01Z","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","version":1},"cited_work":{"arxiv_id":"2404.03413","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.03413","snapshot_observed_at":"2026-07-04T06:39:37.386392Z","title":"Minigpt4-video: Advancing multimodal llms for video understanding with interleaved visual-textual tokens","venue":null,"work_id":"cc937528-86d1-430f-bb5d-4980dbaadd72","year":2024},"citing_paper":{"arxiv_id":"2410.17434","last_updated":"2024-10-22T21:21:37Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-22T21:21:37Z","title":"LongVU: Spatiotemporal Adaptive Compression for Long Video-Language Understanding","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-16T13:53:33.585035Z"},"links":{"cited_paper":"/paper/2404.03413","citing_paper":"/paper/2410.17434"},"observation_digest":"sha256:697ab2e451913a29e2c7eeceddfd8295e822702216d855ba76ddaf53d9dba6cc","observation_id":"d75d13d1-16a3-4873-9478-01a542fae4cd","resolution":{"observed_at":"2026-05-16T13:53:33.610283Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03413","last_updated":"2024-04-04T12:46:01Z","snapshot_observed_at":"2026-07-06T17:55:36.748672Z","submitted_at":"2024-04-04T12:46:01Z","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","version":1},"cited_work":{"arxiv_id":"2404.03413","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.03413","snapshot_observed_at":"2026-07-04T06:39:37.386392Z","title":"Minigpt4-video: Advancing multimodal llms for video understanding with interleaved visual-textual tokens","venue":null,"work_id":"cc937528-86d1-430f-bb5d-4980dbaadd72","year":2024},"citing_paper":{"arxiv_id":"2501.05067","last_updated":"2026-04-20T07:42:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-09T08:43:57Z","title":"LLaVA-Octopus: Unlocking Instruction-Driven Adaptive Projector Fusion for Video Understanding","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-23T06:01:00.775721Z"},"links":{"cited_paper":"/paper/2404.03413","citing_paper":"/paper/2501.05067"},"observation_digest":"sha256:8baa7e6a6b3eec2f3aaa04e047f33e6606f20ace80a6737875ad6840529f0aac","observation_id":"7bccb503-6199-46e9-9652-c5ee814d49e5","resolution":{"observed_at":"2026-05-23T06:02:37.306020Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03413","last_updated":"2024-04-04T12:46:01Z","snapshot_observed_at":"2026-07-06T17:55:36.748672Z","submitted_at":"2024-04-04T12:46:01Z","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","version":1},"cited_work":{"arxiv_id":"2404.03413","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.03413","snapshot_observed_at":"2026-07-04T06:39:37.386392Z","title":"Minigpt4-video: Advancing multimodal llms for video understanding with interleaved visual-textual tokens","venue":null,"work_id":"cc937528-86d1-430f-bb5d-4980dbaadd72","year":2024},"citing_paper":{"arxiv_id":"2501.13106","last_updated":"2025-06-03T03:33:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T18:59:46Z","title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","version":4},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-11T01:19:59.603343Z"},"links":{"cited_paper":"/paper/2404.03413","citing_paper":"/paper/2501.13106"},"observation_digest":"sha256:4fbb91d8c2f6d7eb04316d08022ce3794571e42746c6039858f02d87c079e8c8","observation_id":"d5a8ee7e-ac4b-4091-9af7-9c75d21190cd","resolution":{"observed_at":"2026-05-11T01:20:00.070072Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03413","last_updated":"2024-04-04T12:46:01Z","snapshot_observed_at":"2026-07-06T17:55:36.748672Z","submitted_at":"2024-04-04T12:46:01Z","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.03413","snapshot_observed_at":"2026-08-07T15:29:59.428434Z","title":"Minigpt4-video: Advancing multimodal llms for video understanding with interleaved visual-textual tokens.arXiv preprint arXiv:2404.03413, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17090","last_updated":"2025-05-20T22:14:38Z","snapshot_observed_at":"2026-08-07T15:24:41.300984Z","submitted_at":"2025-05-20T22:14:38Z","title":"EmoSign: A Multimodal Dataset for Understanding Emotions in American Sign Language","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T15:29:59.428434Z"},"links":{"cited_paper":"/paper/2404.03413","citing_paper":"/paper/2505.17090"},"observation_digest":"sha256:6a43fd9641e7276e4c7a89136fabacc3c6abd084986b520ac2d89f1fcd9e0d48","observation_id":"d2f6c73f-2372-44a3-b160-0537fb465047","resolution":{"observed_at":"2026-08-07T15:29:59.428434Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03413","last_updated":"2024-04-04T12:46:01Z","snapshot_observed_at":"2026-07-06T17:55:36.748672Z","submitted_at":"2024-04-04T12:46:01Z","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.03413","snapshot_observed_at":"2026-08-07T14:06:30.173213Z","title":"Minigpt4-video: Advancing multimodal llms for video understanding with interleaved visual-textual tokens","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.20027","last_updated":"2025-05-26T14:17:08Z","snapshot_observed_at":"2026-08-07T13:58:47.137937Z","submitted_at":"2025-05-26T14:17:08Z","title":"Multi-modal brain encoding models for multi-modal stimuli","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-07T14:06:30.173213Z"},"links":{"cited_paper":"/paper/2404.03413","citing_paper":"/paper/2505.20027"},"observation_digest":"sha256:3dff8f8b0d003b03151662075b1d948a75e5f42b94338fcba24e342406ad0dd8","observation_id":"1b4b6cbc-051e-458e-86a3-e7dab822e61a","resolution":{"observed_at":"2026-08-07T14:06:30.173213Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03413","last_updated":"2024-04-04T12:46:01Z","snapshot_observed_at":"2026-07-06T17:55:36.748672Z","submitted_at":"2024-04-04T12:46:01Z","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.03413","snapshot_observed_at":"2026-08-07T12:32:13.398642Z","title":"Minigpt4-video: Advancing multimodal llms for video understanding with interleaved visual-textual tokens.arXiv preprint arXiv:2404.03413, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24346","last_updated":"2025-05-30T08:39:36Z","snapshot_observed_at":"2026-08-07T13:53:39.823176Z","submitted_at":"2025-05-30T08:39:36Z","title":"VUDG: A Dataset for Video Understanding Domain Generalization","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T12:32:13.398642Z"},"links":{"cited_paper":"/paper/2404.03413","citing_paper":"/paper/2505.24346"},"observation_digest":"sha256:c16f170f93d61774ad841123e40b26dd653e52737d1c3ad8d1e64aa9cca07a75","observation_id":"b324e589-68a1-4714-8d08-3deb4559a04d","resolution":{"observed_at":"2026-08-07T12:32:13.398642Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03413","last_updated":"2024-04-04T12:46:01Z","snapshot_observed_at":"2026-07-06T17:55:36.748672Z","submitted_at":"2024-04-04T12:46:01Z","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.03413","snapshot_observed_at":"2026-08-07T13:17:00.125179Z","title":"Minigpt4-video: Advancing multimodal llms for video understanding with interleaved visual-textual tokens","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.00043","last_updated":"2025-05-28T11:21:33Z","snapshot_observed_at":"2026-08-07T13:09:45.267667Z","submitted_at":"2025-05-28T11:21:33Z","title":"From Motion to Behavior: Hierarchical Modeling of Humanoid Generative Behavior Control","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T13:17:00.125179Z"},"links":{"cited_paper":"/paper/2404.03413","citing_paper":"/paper/2506.00043"},"observation_digest":"sha256:91fd54f2e212c29280dc114a580abad91d74accebf4e0dfd833192a609a4a164","observation_id":"b24fe65a-5f5a-4cb0-ba0e-6cc9c46d922e","resolution":{"observed_at":"2026-08-07T13:17:00.125179Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03413","last_updated":"2024-04-04T12:46:01Z","snapshot_observed_at":"2026-07-06T17:55:36.748672Z","submitted_at":"2024-04-04T12:46:01Z","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.03413","snapshot_observed_at":"2026-08-07T10:28:51.919841Z","title":"Minigpt4-video: Advancing multimodal llms for video understanding with interleaved visual-textual tokens.arXiv preprint arXiv:2404.03413, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05260","last_updated":"2025-06-05T17:21:16Z","snapshot_observed_at":"2026-08-07T10:19:37.470353Z","submitted_at":"2025-06-05T17:21:16Z","title":"LeanPO: Lean Preference Optimization for Likelihood Alignment in Video-LLMs","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:51.919841Z"},"links":{"cited_paper":"/paper/2404.03413","citing_paper":"/paper/2506.05260"},"observation_digest":"sha256:30dab0911bbaa8920dc602032cdde4f5a8c2f8a0eae767dcbe842e9956bfc1da","observation_id":"90c069d5-78fa-4889-9b20-ecc8e2e66d6c","resolution":{"observed_at":"2026-08-07T10:28:51.919841Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03413","last_updated":"2024-04-04T12:46:01Z","snapshot_observed_at":"2026-07-06T17:55:36.748672Z","submitted_at":"2024-04-04T12:46:01Z","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.03413","snapshot_observed_at":"2026-08-07T04:43:33.106252Z","title":"Minigpt4-video: Advancing multimodal llms for video understanding with interleaved visual-textual tokens","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.09954","last_updated":"2025-06-11T17:23:41Z","snapshot_observed_at":"2026-08-07T09:15:25.343290Z","submitted_at":"2025-06-11T17:23:41Z","title":"Vision Generalist Model: A Survey","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-07T04:43:33.106252Z"},"links":{"cited_paper":"/paper/2404.03413","citing_paper":"/paper/2506.09954"},"observation_digest":"sha256:23ce5a5477db5a65c19a00fdcd92834d8b98c62a9d45ee646d6c73c73770b967","observation_id":"8be7d67e-4201-4abc-acb6-e92aa4708ba6","resolution":{"observed_at":"2026-08-07T04:43:33.106252Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03413","last_updated":"2024-04-04T12:46:01Z","snapshot_observed_at":"2026-07-06T17:55:36.748672Z","submitted_at":"2024-04-04T12:46:01Z","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.03413","snapshot_observed_at":"2026-08-06T23:49:25.241174Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.16258","last_updated":"2025-06-19T12:17:31Z","snapshot_observed_at":"2026-08-06T23:41:53.635842Z","submitted_at":"2025-06-19T12:17:31Z","title":"ViFusion: In-Network Tensor Fusion for Scalable Video Feature Indexing","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T23:49:25.241174Z"},"links":{"cited_paper":"/paper/2404.03413","citing_paper":"/paper/2506.16258"},"observation_digest":"sha256:a6401a204cfeb9847971566cf79939d4799ee88f886c91e2697b130db8c3436b","observation_id":"95c4507e-4f56-42ef-bd11-4cb240c38ea9","resolution":{"observed_at":"2026-08-06T23:49:25.241174Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03413","last_updated":"2024-04-04T12:46:01Z","snapshot_observed_at":"2026-07-06T17:55:36.748672Z","submitted_at":"2024-04-04T12:46:01Z","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.03413","snapshot_observed_at":"2026-08-06T19:15:51.936727Z","title":"Minigpt4-video: Advancing multimodal llms for video understanding with interleaved visual-textual tokens","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06072","last_updated":"2025-07-08T15:14:53Z","snapshot_observed_at":"2026-08-06T19:09:23.013514Z","submitted_at":"2025-07-08T15:14:53Z","title":"MCAM: Multimodal Causal Analysis Model for Ego-Vehicle-Level Driving Video Understanding","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T19:15:51.936727Z"},"links":{"cited_paper":"/paper/2404.03413","citing_paper":"/paper/2507.06072"},"observation_digest":"sha256:adaae80b971f9f8f301fa5eb9c0fa609a6c763bef9b0f6a0c5809b3201385a0a","observation_id":"33c76992-50e8-44ba-9246-b3c7cc996789","resolution":{"observed_at":"2026-08-06T19:15:51.936727Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03413","last_updated":"2024-04-04T12:46:01Z","snapshot_observed_at":"2026-07-06T17:55:36.748672Z","submitted_at":"2024-04-04T12:46:01Z","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.03413","snapshot_observed_at":"2026-08-05T20:52:11.947587Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.09789","last_updated":"2025-08-13T13:19:31Z","snapshot_observed_at":"2026-08-07T19:48:56.301763Z","submitted_at":"2025-08-13T13:19:31Z","title":"Describe What You See with Multimodal Large Language Models to Enhance Video Recommendations","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-05T20:52:11.947587Z"},"links":{"cited_paper":"/paper/2404.03413","citing_paper":"/paper/2508.09789"},"observation_digest":"sha256:b73d32858fe239d75e58262f4a5f3b43cf0de3bf4fa53711e2b0335a3c2f239c","observation_id":"53afb53d-4905-43c0-bc65-1518f0ee4249","resolution":{"observed_at":"2026-08-05T20:52:11.947587Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03413","last_updated":"2024-04-04T12:46:01Z","snapshot_observed_at":"2026-07-06T17:55:36.748672Z","submitted_at":"2024-04-04T12:46:01Z","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.03413","snapshot_observed_at":"2026-08-05T12:42:48.068160Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.01337","last_updated":"2025-09-01T10:18:47Z","snapshot_observed_at":"2026-08-07T03:15:48.990585Z","submitted_at":"2025-09-01T10:18:47Z","title":"LLM-Guided Semantic Relational Reasoning for Multimodal Intent Recognition","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-05T12:42:48.068160Z"},"links":{"cited_paper":"/paper/2404.03413","citing_paper":"/paper/2509.01337"},"observation_digest":"sha256:cbb796080fe0d1bb9583f24c2f8612e0e25c7ae3f3239a436ba441058bb17d8c","observation_id":"5c40a1ee-0707-4abd-8809-7b85169751e1","resolution":{"observed_at":"2026-08-05T12:42:48.068160Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03413","last_updated":"2024-04-04T12:46:01Z","snapshot_observed_at":"2026-07-06T17:55:36.748672Z","submitted_at":"2024-04-04T12:46:01Z","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","version":1},"cited_work":{"arxiv_id":"2404.03413","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.03413","snapshot_observed_at":"2026-07-04T06:39:37.386392Z","title":"Minigpt4-video: Advancing multimodal llms for video understanding with interleaved visual-textual tokens","venue":null,"work_id":"cc937528-86d1-430f-bb5d-4980dbaadd72","year":2024},"citing_paper":{"arxiv_id":"2511.21998","last_updated":"2026-04-12T05:29:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-11-27T00:54:35Z","title":"Can Multi-Modal LLMs Provide Live Step-by-Step Task Guidance?","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-17T05:36:09.208754Z"},"links":{"cited_paper":"/paper/2404.03413","citing_paper":"/paper/2511.21998"},"observation_digest":"sha256:d4d5dee243118d1d002310369994c088c1bf5a7c1e71b16deec540b069b42b71","observation_id":"f9885207-34f9-4eef-b420-a4429e96f580","resolution":{"observed_at":"2026-05-17T05:39:06.191591Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03413","last_updated":"2024-04-04T12:46:01Z","snapshot_observed_at":"2026-07-06T17:55:36.748672Z","submitted_at":"2024-04-04T12:46:01Z","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","version":1},"cited_work":{"arxiv_id":"2404.03413","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.03413","snapshot_observed_at":"2026-07-04T06:39:37.386392Z","title":"Minigpt4-video: Advancing multimodal llms for video understanding with interleaved visual-textual tokens","venue":null,"work_id":"cc937528-86d1-430f-bb5d-4980dbaadd72","year":2024},"citing_paper":{"arxiv_id":"2604.05079","last_updated":"2026-04-06T18:30:50Z","snapshot_observed_at":"2026-08-02T23:48:40.440218Z","submitted_at":"2026-04-06T18:30:50Z","title":"SVAgent: Storyline-Guided Long Video Understanding via Cross-Modal Multi-Agent Collaboration","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T20:20:08.590407Z"},"links":{"cited_paper":"/paper/2404.03413","citing_paper":"/paper/2604.05079"},"observation_digest":"sha256:ed80a100473faea8b326269874d1ba4708ed6cc12383d8aa7b23ec6122dcc48c","observation_id":"8aba6b87-1732-4856-9e30-2136e5deb11d","resolution":{"observed_at":"2026-05-10T22:00:48.978281Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03413","last_updated":"2024-04-04T12:46:01Z","snapshot_observed_at":"2026-07-06T17:55:36.748672Z","submitted_at":"2024-04-04T12:46:01Z","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","version":1},"cited_work":{"arxiv_id":"2404.03413","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.03413","snapshot_observed_at":"2026-07-04T06:39:37.386392Z","title":"Minigpt4-video: Advancing multimodal llms for video understanding with interleaved visual-textual tokens","venue":null,"work_id":"cc937528-86d1-430f-bb5d-4980dbaadd72","year":2024},"citing_paper":{"arxiv_id":"2604.11283","last_updated":"2026-06-01T08:50:38Z","snapshot_observed_at":"2026-08-02T05:26:21.584719Z","submitted_at":"2026-04-13T10:42:31Z","title":"Multimodal Large Language Model-Enabled Video Translation: A Role-Oriented Survey","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T16:36:33.264166Z"},"links":{"cited_paper":"/paper/2404.03413","citing_paper":"/paper/2604.11283"},"observation_digest":"sha256:29419eea05be0aeaeb340ac8582afe84e520d07ffe4d5b41f8a74e0059d03262","observation_id":"255a21a5-7f8b-4120-acc6-f40f65ea49f1","resolution":{"observed_at":"2026-05-11T08:30:56.941287Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03413","last_updated":"2024-04-04T12:46:01Z","snapshot_observed_at":"2026-07-06T17:55:36.748672Z","submitted_at":"2024-04-04T12:46:01Z","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.03413","snapshot_observed_at":"2026-07-12T22:04:31.302192Z","title":"Minigpt4-video: Advancing multimodal llms for video understanding with interleaved visual-textual tokens,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2604.11283","last_updated":"2026-06-01T08:50:38Z","snapshot_observed_at":"2026-08-02T05:26:21.584719Z","submitted_at":"2026-04-13T10:42:31Z","title":"Multimodal Large Language Model-Enabled Video Translation: A Role-Oriented Survey","version":2},"reference_index":158,"source":"pdf_text","source_observed_at":"2026-07-12T22:04:31.302192Z"},"links":{"cited_paper":"/paper/2404.03413","citing_paper":"/paper/2604.11283"},"observation_digest":"sha256:2fb2f619b76d38ee0368f162b9ca57db013ee67bd3bdca15a23ebf2175033ec2","observation_id":"7a72c2ee-6597-44f7-9e06-bbaebd2e3e40","resolution":{"observed_at":"2026-07-12T22:04:31.302192Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03413","last_updated":"2024-04-04T12:46:01Z","snapshot_observed_at":"2026-07-06T17:55:36.748672Z","submitted_at":"2024-04-04T12:46:01Z","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","version":1},"cited_work":{"arxiv_id":"2404.03413","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.03413","snapshot_observed_at":"2026-07-04T06:39:37.386392Z","title":"Minigpt4-video: Advancing multimodal llms for video understanding with interleaved visual-textual tokens","venue":null,"work_id":"cc937528-86d1-430f-bb5d-4980dbaadd72","year":2024},"citing_paper":{"arxiv_id":"2604.14149","last_updated":"2026-04-16T15:48:38Z","snapshot_observed_at":"2026-07-06T23:02:00.082783Z","submitted_at":"2026-04-15T17:59:52Z","title":"One Token per Highly Selective Frame: Towards Extreme Compression for Long Video Understanding","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T13:28:58.920442Z"},"links":{"cited_paper":"/paper/2404.03413","citing_paper":"/paper/2604.14149"},"observation_digest":"sha256:fde2039bc1a7c90a7caeffc3b81609fc4141090e9fd7ac89956bc456cf9119e7","observation_id":"660701bf-6df2-4b57-801e-7e7628952385","resolution":{"observed_at":"2026-05-10T13:30:26.574338Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03413","last_updated":"2024-04-04T12:46:01Z","snapshot_observed_at":"2026-07-06T17:55:36.748672Z","submitted_at":"2024-04-04T12:46:01Z","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","version":1},"cited_work":{"arxiv_id":"2404.03413","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.03413","snapshot_observed_at":"2026-07-04T06:39:37.386392Z","title":"Minigpt4-video: Advancing multimodal llms for video understanding with interleaved visual-textual tokens","venue":null,"work_id":"cc937528-86d1-430f-bb5d-4980dbaadd72","year":2024},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-07-06T23:46:42.693840Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2404.03413","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:e592ce2093041f1ba0d45dcc6bbe321e94bd6549e50cd975135c2895b458f056","observation_id":"e44ff72d-fe3d-44eb-90f9-764ea821fa25","resolution":{"observed_at":"2026-07-02T17:07:12.876383Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03413","last_updated":"2024-04-04T12:46:01Z","snapshot_observed_at":"2026-07-06T17:55:36.748672Z","submitted_at":"2024-04-04T12:46:01Z","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","version":1},"cited_work":{"arxiv_id":"2404.03413","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.03413","snapshot_observed_at":"2026-07-04T06:39:37.386392Z","title":"Minigpt4-video: Advancing multimodal llms for video understanding with interleaved visual-textual tokens","venue":null,"work_id":"cc937528-86d1-430f-bb5d-4980dbaadd72","year":2024},"citing_paper":{"arxiv_id":"2606.17798","last_updated":"2026-06-16T11:18:05Z","snapshot_observed_at":"2026-07-06T23:53:21.186284Z","submitted_at":"2026-06-16T11:18:05Z","title":"LiveStarPro: Proactive Streaming Video Understanding with Hierarchical Memory for Long-Horizon Streams","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-27T01:12:46.295455Z"},"links":{"cited_paper":"/paper/2404.03413","citing_paper":"/paper/2606.17798"},"observation_digest":"sha256:92919b29d014e03095af4ae8534786d547b022fc4b10df0c822502648c300d8d","observation_id":"850ea5ce-203f-4cd4-ac50-d4befa26a7f0","resolution":{"observed_at":"2026-07-03T20:38:56.192029Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03413","last_updated":"2024-04-04T12:46:01Z","snapshot_observed_at":"2026-07-06T17:55:36.748672Z","submitted_at":"2024-04-04T12:46:01Z","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","version":1},"cited_work":{"arxiv_id":"2404.03413","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.03413","snapshot_observed_at":"2026-07-04T06:39:37.386392Z","title":"Minigpt4-video: Advancing multimodal llms for video understanding with interleaved visual-textual tokens","venue":null,"work_id":"cc937528-86d1-430f-bb5d-4980dbaadd72","year":2024},"citing_paper":{"arxiv_id":"2606.21734","last_updated":"2026-06-19T20:43:49Z","snapshot_observed_at":"2026-08-05T18:05:51.515234Z","submitted_at":"2026-06-19T20:43:49Z","title":"HPP: Hierarchical Programmatic Probing for Long Video Understanding by Decoupling Perception and Reasoning","version":1},"reference_index":147,"source":"arxiv_source","source_observed_at":"2026-06-26T14:19:53.450263Z"},"links":{"cited_paper":"/paper/2404.03413","citing_paper":"/paper/2606.21734"},"observation_digest":"sha256:f2b5b425b1e7226d6e1ee0bf3425f0c689b347e84bed81fb2a1091dfa814034c","observation_id":"670b309a-0678-4391-96da-8b21ed8b59ad","resolution":{"observed_at":"2026-07-04T06:39:37.387790Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03413","last_updated":"2024-04-04T12:46:01Z","snapshot_observed_at":"2026-07-06T17:55:36.748672Z","submitted_at":"2024-04-04T12:46:01Z","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","version":1},"cited_work":{"arxiv_id":"2404.03413","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.03413","snapshot_observed_at":"2026-07-04T06:39:37.386392Z","title":"Minigpt4-video: Advancing multimodal llms for video understanding with interleaved visual-textual tokens","venue":null,"work_id":"cc937528-86d1-430f-bb5d-4980dbaadd72","year":2024},"citing_paper":{"arxiv_id":"2606.30288","last_updated":"2026-06-29T13:30:17Z","snapshot_observed_at":"2026-07-07T00:04:10.492340Z","submitted_at":"2026-06-29T13:30:17Z","title":"VisReflect: Latent Visual Reflection for Fine-Grained Perception in Long Visual Context","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-30T06:01:49.904752Z"},"links":{"cited_paper":"/paper/2404.03413","citing_paper":"/paper/2606.30288"},"observation_digest":"sha256:9200043504ff9fffc663cc4190b1d9433cb340a524f9cb970ee67a330e1d7e34","observation_id":"b7bb3f1d-09d1-4fc2-bb2b-9f0d9867f14d","resolution":{"observed_at":"2026-06-30T06:04:21.289121Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03413","last_updated":"2024-04-04T12:46:01Z","snapshot_observed_at":"2026-07-06T17:55:36.748672Z","submitted_at":"2024-04-04T12:46:01Z","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.03413","snapshot_observed_at":"2026-07-11T08:59:46.244502Z","title":"arXiv preprint arXiv:2404.03413 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.05089","last_updated":"2026-07-06T13:50:15Z","snapshot_observed_at":"2026-07-11T08:59:45.751461Z","submitted_at":"2026-07-06T13:50:15Z","title":"TimeThink: Reasoning with Time for Video LLMs","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-11T08:59:46.244502Z"},"links":{"cited_paper":"/paper/2404.03413","citing_paper":"/paper/2607.05089"},"observation_digest":"sha256:b273d404f432086f52eeac4dce0adca23d8787010b918b63d1487d2e2e091998","observation_id":"20a41ddb-143e-4c2f-a16e-6dcf548dd89c","resolution":{"observed_at":"2026-07-11T08:59:46.244502Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03413","last_updated":"2024-04-04T12:46:01Z","snapshot_observed_at":"2026-07-06T17:55:36.748672Z","submitted_at":"2024-04-04T12:46:01Z","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.03413","snapshot_observed_at":"2026-08-01T00:45:40.479683Z","title":"arXiv preprint arXiv:2404.03413 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.27826","last_updated":"2026-07-30T08:05:26Z","snapshot_observed_at":"2026-08-04T13:04:08.883511Z","submitted_at":"2026-07-30T08:05:26Z","title":"Sign Language Question Answering: A New Task, Benchmark, and Baseline for Sign Language Understanding","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-01T00:45:40.479683Z"},"links":{"cited_paper":"/paper/2404.03413","citing_paper":"/paper/2607.27826"},"observation_digest":"sha256:5bce6cd0a6fc0080c7311b3c77abb1fd3453408e2ba0447ebf624ac60f333590","observation_id":"b5bbbc41-1e21-43ea-bae8-2169698fb5a4","resolution":{"observed_at":"2026-08-01T00:45:40.479683Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03413","last_updated":"2024-04-04T12:46:01Z","snapshot_observed_at":"2026-07-06T17:55:36.748672Z","submitted_at":"2024-04-04T12:46:01Z","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.03413","snapshot_observed_at":"2026-07-31T11:36:26.082445Z","title":"arXiv preprint arXiv:2404.03413 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.28312","last_updated":"2026-08-01T16:00:11Z","snapshot_observed_at":"2026-08-06T23:11:29.524233Z","submitted_at":"2026-07-30T14:47:00Z","title":"ObjectStream: Latent Objects as Memory Anchors for Streaming Video Understanding","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-07-31T11:36:26.082445Z"},"links":{"cited_paper":"/paper/2404.03413","citing_paper":"/paper/2607.28312"},"observation_digest":"sha256:d4a9d9eace37c6be1e1f67060fd42807ca3d353801d96df0eec24eba0c9f1b95","observation_id":"556337f2-1942-4cbc-93e3-483748e20b39","resolution":{"observed_at":"2026-07-31T11:36:26.082445Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2404.03413/citation-record","integrity":"/paper/2404.03413/integrity","json":"/paper/2404.03413/citation-record.json","paper":"/paper/2404.03413"},"outbound":[],"paper":{"arxiv_id":"2404.03413","last_updated":"2024-04-04T12:46:01Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T17:55:36.748672Z","submitted_at":"2024-04-04T12:46:01Z","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 27 inbound Pith citation observations for arXiv:2404.03413."}