{"as_of":"2026-08-16T23:36:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:114f55035a32472d86cbddff97ade63e9dd575a27a59d3223d4ed0352c107d5b","coverage":[{"denominator":84,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":84,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T12:37:24.588694Z","state":"measured"},{"denominator":87,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":87,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-16T06:30:59.297886+00:00","state":"measured"},{"denominator":3,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":3,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T14:39:08.699220Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-12T03:06:18.765685Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.24158","snapshot_observed_at":"2026-08-04T13:27:56.359929Z","title":"Threading keyframe with narratives: Mllms as strong long video comprehenders.arXiv preprint arXiv:2505.24158,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2510.00705","last_updated":"2026-06-26T15:22:20Z","snapshot_observed_at":"2026-08-14T01:20:42.424375Z","submitted_at":"2025-10-01T09:20:51Z","title":"Training-free Uncertainty Guidance for Complex Visual Tasks with MLLMs","version":3},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-04T13:27:56.359929Z"},"links":{"cited_paper":"/paper/2505.24158","citing_paper":"/paper/2510.00705"},"observation_digest":"sha256:c56dc8a3d58a124ac37ea7d835979ace758c78076e031a707af055607d4370ca","observation_id":"76dbb3a0-84e8-434c-9c6f-48b3029920af","resolution":{"observed_at":"2026-08-04T13:27:56.359929Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"cited_work":{"arxiv_id":"2505.24158","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.24158","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"b67f3704-b24e-466c-96b8-410615681160","year":null},"citing_paper":{"arxiv_id":"2605.09223","last_updated":"2026-07-30T19:00:53Z","snapshot_observed_at":"2026-08-10T22:41:15.175214Z","submitted_at":"2026-05-09T23:47:46Z","title":"CREST: Curvature-Regulated Event-Centric Sampling for Efficient Long-Video Understanding","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-12T03:06:09.753634Z"},"links":{"cited_paper":"/paper/2505.24158","citing_paper":"/paper/2605.09223"},"observation_digest":"sha256:f51d7b09ae87863a692161e1553ba2340a6cc447805a7a85100ac874090917fd","observation_id":"767c7c7e-b77e-4aeb-8c50-e2dfe77869ac","resolution":{"observed_at":"2026-05-12T03:06:18.772638Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.24158","snapshot_observed_at":"2026-08-15T14:39:08.699220Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.05707","last_updated":"2026-08-06T07:47:15Z","snapshot_observed_at":"2026-08-16T18:51:19.932121Z","submitted_at":"2026-08-06T07:47:15Z","title":"One Ranking, Any Budget: Matryoshka Evidence-to-Context Frame Selection for Long-Video Understanding","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-15T14:39:08.699220Z"},"links":{"cited_paper":"/paper/2505.24158","citing_paper":"/paper/2608.05707"},"observation_digest":"sha256:167d980543e97175e18ddd085ff3bc0632adbb75fa07351c2d71f81e8fa2ceb9","observation_id":"df5b6b56-6f9f-4886-9414-484068385707","resolution":{"observed_at":"2026-08-15T14:39:08.699220Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2505.24158/citation-record","integrity":"/paper/2505.24158/integrity","json":"/paper/2505.24158/citation-record.json","paper":"/paper/2505.24158"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-16T19:40:28.523700Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-07T12:37:16.067831Z","title":"Gpt-4 technical report","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:16.067831Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:1850cb3f9f33354c9484c7872eed73c0ae66c2c4bde8dd379e62f927cb76663a","observation_id":"0b0f98ab-5ecc-4387-99bc-7583abcaa79a","resolution":{"observed_at":"2026-08-07T12:37:16.067831Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:16.151840Z","title":"Flamingo: a visual language model for few-shot learning","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:16.151840Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:662b390c60a5a78f0f2778db30e96ea9b9f54520e03a9e6ef50be4380751d1f8","observation_id":"2a25c717-9def-4308-8ff1-c6049a818e78","resolution":{"observed_at":"2026-08-07T12:37:16.151840Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:16.236696Z","title":"Vqa: Visual question answering","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:16.236696Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:d347e738c4b33eb2ebb8611ca9b41e80c4fd11b3b340d3c99a090ef7cfa9fc41","observation_id":"68155e93-d9bf-49c6-b5da-6cad2496cd05","resolution":{"observed_at":"2026-08-07T12:37:16.236696Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12966","last_updated":"2023-10-13T02:41:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-24T17:59:17Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12966","snapshot_observed_at":"2026-08-07T12:37:16.323580Z","title":"Qwen-vl: A frontier large vision-language model with versatile abilities","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:16.323580Z"},"links":{"cited_paper":"/paper/2308.12966","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:fe1cb8d5c9c7d9a6612c3c0d6a12464ca7da7d3206af5d69870d376c0238c51c","observation_id":"5bc3e1f1-f3a4-4f92-a82e-04cd9341d2bc","resolution":{"observed_at":"2026-08-07T12:37:16.323580Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:16.421314Z","title":"Solving mixed-integer quadratic programming problems with ibm-cplex: a progress report","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:16.421314Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:d42ad5721cc6eb094284d28060814bcdc41c4dfa5614e5f0c0771067198b2012","observation_id":"e7c4e58b-8675-472d-a7ed-ac3f59df9c45","resolution":{"observed_at":"2026-08-07T12:37:16.421314Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2108.07258","last_updated":"2022-07-12T23:45:14Z","snapshot_observed_at":"2026-08-02T09:20:40.804790Z","submitted_at":"2021-08-16T17:50:08Z","title":"On the Opportunities and Risks of Foundation Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2108.07258","snapshot_observed_at":"2026-08-07T12:37:16.506733Z","title":"On the opportunities and risks of foundation models","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:16.506733Z"},"links":{"cited_paper":"/paper/2108.07258","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:ccc556a480ae74d8b27895fd30389ab665b1253f192ab7da0305066463c78b6d","observation_id":"02f8b0ea-6d17-4e2d-9aba-f79e9be26d52","resolution":{"observed_at":"2026-08-07T12:37:16.506733Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:16.620624Z","title":"Language models are few-shot learners","venue":null,"work_id":null,"year":1901},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:16.620624Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:5e531b129b717005fbbd4d7893b222a6d994be7b586f4fa22f147ca13a03994d","observation_id":"8b1037ce-82e4-47c4-a89d-2e18f8978df0","resolution":{"observed_at":"2026-08-07T12:37:16.620624Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:34.412934Z","title":"Hourvideo: 1-hour video-language understanding","venue":null,"work_id":"fe8e8eaf-216c-476b-bd20-d2412037a1da","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:16.681958Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:0a66f636130075bde64c7dd50c01d18ec4b5b2b76f279b7b759e83e3d97391f2","observation_id":"896ac7d1-7c57-4717-a50c-28a592b161b9","resolution":{"observed_at":"2026-08-07T12:37:34.498288Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:34.269421Z","title":"Sharegpt4video: Improving video understanding and generation with better captions","venue":null,"work_id":"ed43e894-b84d-4cb6-bf04-4beeb0c04da3","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:16.776741Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:3ae7105e2ce8efe084e72d50987d77a3f98153fb2cc050a0d5ce5baee7f3ae22","observation_id":"c3568a3a-b413-41e3-88e1-e0f954ffb8ae","resolution":{"observed_at":"2026-08-07T12:37:34.329297Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.10188","last_updated":"2024-12-13T02:32:06Z","snapshot_observed_at":"2026-08-16T04:27:36.181491Z","submitted_at":"2024-08-19T17:48:08Z","title":"LongVILA: Scaling Long-Context Visual Language Models for Long Videos","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.10188","snapshot_observed_at":"2026-08-07T12:37:16.941581Z","title":"Longvila: Scaling long-context visual language models for long videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:16.941581Z"},"links":{"cited_paper":"/paper/2408.10188","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:3381b869fd4eb55df461b0a8dc900f3ae5d5488191bcd93e1ef7d775c1393b0f","observation_id":"27e4ed29-6118-42b5-b76a-559978453cd2","resolution":{"observed_at":"2026-08-07T12:37:16.941581Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:34.019636Z","title":"Internvl: Scaling up vision foundation models and aligning for generic visual-linguistic tasks","venue":null,"work_id":"6b7b65c2-4028-477d-b1d4-701b1752ba2d","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:17.185352Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:7f0ddaf90f5500c79697af77b6ac245e97a879cdc11ab940754f86ca412fe46b","observation_id":"e3b44931-b510-4d07-8757-d9f2446ea122","resolution":{"observed_at":"2026-08-07T12:37:34.139239Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07476","last_updated":"2024-10-30T06:49:54Z","snapshot_observed_at":"2026-08-14T16:25:22.654846Z","submitted_at":"2024-06-11T17:22:23Z","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.07476","snapshot_observed_at":"2026-08-07T12:37:17.344219Z","title":"Videollama 2: Advancing spatial-temporal modeling and audio understanding in video-llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:17.344219Z"},"links":{"cited_paper":"/paper/2406.07476","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:79f5222f2433a6e572b8f753d117a0cc42b1d3fa59bb8e3c2dcbe0e66a8177e2","observation_id":"762e141e-19cd-44e0-bc51-3a40d50f8291","resolution":{"observed_at":"2026-08-07T12:37:17.344219Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:33.841826Z","title":"Patch n’pack: Navit, a vision transformer for any aspect ratio and resolution","venue":null,"work_id":"5cef66cc-8b6d-4eec-a4a2-95bcea1d1357","year":2023},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:17.455893Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:2865996264560b39a4b3ae3a22e9116521a364a39b7c23c553aa4eb8b3747bc8","observation_id":"d484031f-54be-415b-940f-bb6fb78340da","resolution":{"observed_at":"2026-08-07T12:37:33.935943Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:33.677794Z","title":"An image is worth 16x16 words: Transformers for image recognition at scale","venue":null,"work_id":"52f61c47-fac0-4f1a-8b18-3dbec1d02387","year":2020},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:17.620244Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:128f6416d8a91420166e2114a5361879921b77f5f94de5b70ec42a9b02cc38b4","observation_id":"376adc86-d204-4089-8f9e-cecd44d61290","resolution":{"observed_at":"2026-08-07T12:37:33.752601Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:33.520806Z","title":"Vlmevalkit: An open-source toolkit for evaluating large multi- modality models","venue":null,"work_id":"9d929e34-e0d1-4fe5-be3f-04702e669ff8","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:17.746820Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:93129806762ebf015606750e95b11a37a3030c76583c370f3e61fffeadda1a8b","observation_id":"2676de86-cf63-4741-8c40-e4de5cae84b2","resolution":{"observed_at":"2026-08-07T12:37:33.587582Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:33.336676Z","title":"Slowfast networks for video recognition","venue":null,"work_id":"c62e5db3-5ef8-4aa7-b3ec-b5e2995772de","year":2019},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:17.902941Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:e48c5b2b9f7e26edca319a9ff22e4311352ee87cd7241aed04e45a6c2741b79a","observation_id":"7f21be05-08c7-49b4-9be9-780019e38bd2","resolution":{"observed_at":"2026-08-07T12:37:33.431518Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:33.064148Z","title":"Video-mme: The first-ever comprehensive evaluation benchmark of multi-modal llms in video analysis","venue":null,"work_id":"5ebd45e1-dc25-4149-836d-79b629b261f4","year":2025},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:18.023177Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:fd4db21d143374d5b1f7460ed1a88439dcfaae5ab6fb98b34cd3a66356442f8d","observation_id":"809b72ba-d3a4-41c5-8700-937a966fc819","resolution":{"observed_at":"2026-08-07T12:37:33.155237Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-07T12:37:18.121807Z","title":"The llama 3 herd of models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:18.121807Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:0d558775a5b65be4d8813cb987a46941932956ecdbdf596756a994206c4d2f2e","observation_id":"47dd4bc0-84f5-4db0-b975-1d6a06d65372","resolution":{"observed_at":"2026-08-07T12:37:18.121807Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:32.876865Z","title":"M-llm based video frame selection for efficient video understanding","venue":null,"work_id":"156e9f8c-38be-4ee8-b1af-b148f01e6fe2","year":2025},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:18.286872Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:fbf9c68ccc1d5ac1ad58b0452bf79522a7364a2502e3d963a7187d3f0cc0f61a","observation_id":"2381eef4-f583-4f76-94e7-c73d273512bf","resolution":{"observed_at":"2026-08-07T12:37:32.959706Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:32.701466Z","title":"Chat-univi: Unified visual representation empowers large language models with image and video understanding","venue":null,"work_id":"250a8329-1659-42e8-bb66-095a8882e5a4","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:18.409318Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:524fedaad01e90298d32db304586278fbf80d70266b391b72103556d5e60a8e8","observation_id":"3750f7a5-fdfc-4ed9-aafa-7789b8c5546b","resolution":{"observed_at":"2026-08-07T12:37:32.786989Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:32.516316Z","title":"Language repository for long video understanding","venue":null,"work_id":"667c4a48-df45-404d-8160-96ebe5204d8b","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:18.555490Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:64f41238aa27ffe31f1ddd2067ad47a66b5da1b98c17cb89524deb3e951377f2","observation_id":"fa668bc0-cee4-43e7-b805-46e964487e50","resolution":{"observed_at":"2026-08-07T12:37:32.625784Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:32.329054Z","title":"An image grid can be worth a video: Zero-shot video question answering using a vlm","venue":null,"work_id":"3a02dfd1-42b5-466e-991f-532ce31977bf","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:18.682198Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:54494e2fe7540d375c684171afb20532254071b3c64cc14b724dba0fa3022773","observation_id":"a8d39fa3-b1cf-4f27-b2a2-2587ffad92a7","resolution":{"observed_at":"2026-08-07T12:37:32.441482Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:32.078456Z","title":"Lmms-eval: Accelerating the development of large multimoal models, March 2024","venue":null,"work_id":"7920c36b-12a4-4027-8985-9dc714bcc882","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:18.798283Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:470ac335efc4f8f5d3fd79760194334a2d2ef78d108495451a94106485d22f3a","observation_id":"4241b8f0-2af6-46c9-9a51-1bf2b12aaab9","resolution":{"observed_at":"2026-08-07T12:37:32.178385Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03326","last_updated":"2024-10-26T16:35:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:59:44Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03326","snapshot_observed_at":"2026-08-07T12:37:18.912546Z","title":"Llava-onevision: Easy visual task transfer","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:18.912546Z"},"links":{"cited_paper":"/paper/2408.03326","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:01f9a847743a5f78d208b12f5c478b1c40345e0c1627dbe1b0ac85f1d0c83a9f","observation_id":"291baed7-2303-4cf2-977a-841f9a3c3bf3","resolution":{"observed_at":"2026-08-07T12:37:18.912546Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:31.857768Z","title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","venue":null,"work_id":"be444341-5ef6-4a05-ae8e-ab7522b5f8eb","year":2023},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:18.979088Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:cd790a1ef83cabb030f3eb1e8956832552c72e28a0f359a6531baf2d15c40dd8","observation_id":"5d203498-19e8-4011-bb17-7995405bec65","resolution":{"observed_at":"2026-08-07T12:37:31.953262Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-07T12:37:19.138328Z","title":"Videochat: Chat-centric video understanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:19.138328Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:06e2c832399e9ca333ebfd1e4ce78827c8126f64e4c37528b57ef3462eedb4ab","observation_id":"60b66c9c-95ff-4c99-aa4f-806c7132fb93","resolution":{"observed_at":"2026-08-07T12:37:19.138328Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:31.707621Z","title":"Llama-vid: An image is worth 2 tokens in large language models","venue":null,"work_id":"5c185666-bbbe-4b66-9fb2-8590fea490e5","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:19.222339Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:3bb36b2f57a8afe6009130eb0635036da4b9d788c4e59209e7891198edb32156","observation_id":"01bfaaa7-1cec-48f4-8706-3dc35bce83d9","resolution":{"observed_at":"2026-08-07T12:37:31.759563Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:31.497843Z","title":"Video-llava: Learning united visual representation by alignment before projection","venue":null,"work_id":"0ef7b386-628b-425c-b049-b9a51695eb70","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:19.291562Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:5f9cdeab9883241c1d3d034288789b20adc641012f6529d85a3442cb3ac6685c","observation_id":"5d5a836a-f36a-4515-98b8-f0983000e696","resolution":{"observed_at":"2026-08-07T12:37:31.614389Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:31.334767Z","title":"Vila: On pre- training for visual language models","venue":null,"work_id":"e1848844-5ed4-4aba-bf93-5fee15f54072","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:19.353735Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:dea57f539b698e3664a47f8f9f0a5f16c57df5a5cc018d5fff1088a6091a2ae4","observation_id":"638b84e3-ccde-4770-8b05-ab1b71fd4652","resolution":{"observed_at":"2026-08-07T12:37:31.426866Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:19.424939Z","title":"Llava-next: Improved reasoning, ocr, and world knowledge, January 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:19.424939Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:be076bd9fdc68c89aeb49b3c7e27ea0da6a434a66466874facf9291ada2d889f","observation_id":"c00adb3d-44ed-426f-a8e7-1acd59c1535a","resolution":{"observed_at":"2026-08-07T12:37:19.424939Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:31.061164Z","title":"Visual instruction tuning.Advances in neural information processing systems (NeurIPS), 36:34892–34916, 2023","venue":null,"work_id":"17d1035b-45dd-4c27-a01c-960ab99edec0","year":2023},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:19.493444Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:c467320e2c4f4a4f628cf64a4f31245557ee744597c95d8d1d42ca963280fdf1","observation_id":"e16f8ea2-1456-4418-80ed-0254c24804d7","resolution":{"observed_at":"2026-08-07T12:37:31.189946Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:30.822419Z","title":"Timecraft: Navigate weakly-supervised temporal grounded video question answering via bi-directional reasoning","venue":null,"work_id":"6430c44f-08e0-4e2d-9cbe-3d5444d8f9cb","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:19.566948Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:bdc246bb39331699196e73b0b056ed3679ebd4408712b47be4cd4677b8a2f355","observation_id":"6954443f-d472-471a-ba74-4e194b2f62cf","resolution":{"observed_at":"2026-08-07T12:37:30.918499Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:30.551095Z","title":"Lost in the middle: How language models use long contexts","venue":null,"work_id":"6c82aa6c-70a4-41be-bffa-a69607b92a06","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:19.644740Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:14b15e38e2a1ce09cd27f883d93c76e8c1b994fc0b4e5da5d6a59b9e824893fe","observation_id":"b426fb8e-9c4a-44d3-9d07-e7078607e366","resolution":{"observed_at":"2026-08-07T12:37:30.659000Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:30.243332Z","title":"St-llm: Large language models are effective temporal learners","venue":null,"work_id":"6bee1e60-3096-4163-90f0-a900a342d4e8","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:19.721077Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:ee2e22ba6d34064b516960f7bff5509a62ba33bfb7bcac08d570a2e8edebe567","observation_id":"2f4cc9f1-282b-41ed-9b30-625553421422","resolution":{"observed_at":"2026-08-07T12:37:30.370543Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:30.052866Z","title":"Bolt: Boost large vision-language model without training for long-form video understanding","venue":null,"work_id":"78ac377d-70ef-47a7-a91a-72aa6d619c28","year":2025},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:19.791097Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:c32b2a878e8c1e488e3894e77acd0a09d31ea68d2c125ac1bb74cf368009690b","observation_id":"81f55e11-6d72-4da8-a94c-73e27f7c50ec","resolution":{"observed_at":"2026-08-07T12:37:30.122315Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:29.895831Z","title":"Drvideo: Document retrieval based long video understanding","venue":null,"work_id":"7641425e-56db-4c8b-b561-b0293420b35d","year":2025},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:19.859288Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:b44eb0d01fced5a32d995df034dec9a3592782875c42f0b7ed84fb009f740027","observation_id":"5342c0cd-7585-46c2-a601-c2740e0a62c8","resolution":{"observed_at":"2026-08-07T12:37:29.987173Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05424","last_updated":"2024-06-10T01:36:53Z","snapshot_observed_at":"2026-07-06T15:40:24.127663Z","submitted_at":"2023-06-08T17:59:56Z","title":"Video-ChatGPT: Towards Detailed Video Understanding via Large Vision and Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05424","snapshot_observed_at":"2026-08-07T12:37:19.932480Z","title":"Video-chatgpt: Towards detailed video understanding via large vision and language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:19.932480Z"},"links":{"cited_paper":"/paper/2306.05424","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:293987a62d44918250f90b5d363dd37a661c32c4a4d326be49a9a3d461193ea8","observation_id":"d4d72409-db44-44cc-b222-457dff76e831","resolution":{"observed_at":"2026-08-07T12:37:19.932480Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:29.695081Z","title":"Egoschema: A diagnostic benchmark for very long-form video language understanding","venue":null,"work_id":"c88238c2-80c7-4846-9256-9932780a9ab7","year":2023},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:20.004759Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:a6fdeb5a7bfe6b6cd0d657e3225df9b2938e1b3c9bad21c76c4391ed4a6b4128","observation_id":"ffbf24f6-cd17-462d-8bb0-3c82bb673924","resolution":{"observed_at":"2026-08-07T12:37:29.773619Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:29.568181Z","title":"Morevqa: Exploring modular reasoning models for video question answering","venue":null,"work_id":"7fde5000-9252-4316-ac61-6b80e4b3b49b","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:20.072028Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:513aff4b744e909bb789ddbd243ba59013fe142214d67d15a2e3eadb2d3e3fc6","observation_id":"ae471a22-25b4-492d-a176-bca78cc195e3","resolution":{"observed_at":"2026-08-07T12:37:29.610912Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:29.450488Z","title":"Branch-and-bound algorithms: A survey of recent advances in searching, branching, and pruning","venue":null,"work_id":"e07157e8-b9c6-463b-b25d-c4069160834a","year":2016},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:20.145930Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:bf4546881f15f17348394aa694c0159ac06fbf1f13983d7417e0c51600e259f1","observation_id":"0abc113c-9916-4985-825b-dc805945947b","resolution":{"observed_at":"2026-08-07T12:37:29.526632Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:29.333001Z","title":"Chatgpt: Optimizing language models for dialogue, 2023","venue":null,"work_id":"87c4749c-cdd7-4bbe-b611-d31ccb3afcee","year":2023},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:20.236870Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:4617a8011adc265073a01bf0f47e227383ebe44e21b5a569c0ed12512a61c572","observation_id":"d1c22e62-5148-4f3c-bd01-6c185a24dec6","resolution":{"observed_at":"2026-08-07T12:37:29.389856Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:29.182548Z","title":"Too many frames, not all useful: Efficient strategies for long-form video qa","venue":null,"work_id":"7691b435-ef1e-4b2c-b9a9-ab727b7df31a","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:20.321049Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:4a420a56b94e2f0c7034cdc1bd3abfa34b93c60190ac0a38e1fa8f16c251cea9","observation_id":"02da2d3a-496d-4709-bcc1-4a52e6737496","resolution":{"observed_at":"2026-08-07T12:37:29.235070Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:29.019542Z","title":"Momentor: Advancing video large language model with fine-grained temporal reasoning","venue":null,"work_id":"d6d73ae9-bfa7-442d-befe-f5508cd53c08","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:20.447778Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:a465b250dab6bad95cadedd6f47c4d90a87e7743d23f5efa63ce786a7009e77f","observation_id":"deb2fcb6-9589-4fa4-b275-b8777c51689d","resolution":{"observed_at":"2026-08-07T12:37:29.090406Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:28.908519Z","title":"Learning transferable visual models from natural language supervision","venue":null,"work_id":"7c53ddfe-fae8-40ed-a208-73f0f0f76692","year":2021},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:20.542794Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:bec6a9a800299e333dc1d1b0f165877c19af210a400582a581dbcaa1c2b65a38","observation_id":"9347cb9c-82d3-4a9e-bf0e-53f2dcf2f7f3","resolution":{"observed_at":"2026-08-07T12:37:28.954666Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:20.635358Z","title":"The knapsack problem: a survey","venue":null,"work_id":null,"year":1975},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:20.635358Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:a21d57275f3058b715887593478ab5bd18913c6b15438d704beb440c2ba6ba09","observation_id":"3bc64dc1-212e-4d68-9cf6-0e9ba1c2a852","resolution":{"observed_at":"2026-08-07T12:37:20.635358Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.17434","last_updated":"2024-10-22T21:21:37Z","snapshot_observed_at":"2026-08-13T11:36:11.821356Z","submitted_at":"2024-10-22T21:21:37Z","title":"LongVU: Spatiotemporal Adaptive Compression for Long Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.17434","snapshot_observed_at":"2026-08-07T12:37:20.727692Z","title":"Longvu: Spatiotemporal adaptive compression for long video-language understanding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:20.727692Z"},"links":{"cited_paper":"/paper/2410.17434","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:d0fd613ce232310b783c2e408274b4512366a03a1af0e896d4a993762d84e2b2","observation_id":"312e0497-9293-483e-9372-348a673c1466","resolution":{"observed_at":"2026-08-07T12:37:20.727692Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.14485","last_updated":"2024-12-10T12:45:31Z","snapshot_observed_at":"2026-08-16T13:16:29.857447Z","submitted_at":"2024-09-22T15:13:31Z","title":"Video-XL: Extra-Long Vision Language Model for Hour-Scale Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.14485","snapshot_observed_at":"2026-08-07T12:37:20.806587Z","title":"Video-xl: Extra-long vision language model for hour-scale video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:20.806587Z"},"links":{"cited_paper":"/paper/2409.14485","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:c5a59fd9756d54b4e1cb0150afafd0b507d6426e799b56d61ed87409f38891c8","observation_id":"f19b5cbc-5429-425a-8010-6d84e9d4ed61","resolution":{"observed_at":"2026-08-07T12:37:20.806587Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:28.755969Z","title":"Two-stream convolutional networks for action recognition in videos","venue":null,"work_id":"74aab278-409e-421b-ae17-7b205cf21b64","year":2014},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:20.873682Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:7b51045458f6d735ac9f4d703e4f2ce47352dba63e33eeae01f9f15a22fe7487","observation_id":"fbd4598c-ac5d-4e0c-aeba-4bd4cc9806cf","resolution":{"observed_at":"2026-08-07T12:37:28.843270Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:28.636677Z","title":"Moviechat: From dense token to sparse memory for long video understanding","venue":null,"work_id":"8530be3e-3582-432a-aad0-f04e6396d29c","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:20.984827Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:6ac7a567e6b673734927147aadebd8ea3f15294bf5e1abc8dffcc52959c849f5","observation_id":"d462dcf3-d607-4bff-9bed-dd3901c0c289","resolution":{"observed_at":"2026-08-07T12:37:28.689205Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:21.085807Z","title":"Mdp3: A training-free approach for list-wise frame selection in video-llms","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:21.085807Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:44f43c339d4d87207a6a103a4d5f71916291a2fe5123afeef2490772e1507958","observation_id":"7b755651-2206-48b2-9d89-86b81dba6344","resolution":{"observed_at":"2026-08-07T12:37:21.085807Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:28.454405Z","title":"Adaptive keyframe sampling for long video understanding","venue":null,"work_id":"4df9b1cf-8b45-4483-8d66-f457e90677d7","year":2025},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:21.159282Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:3c47a3df7a69db519f5f47e1310df0d93ec49fde6987cb604df20cd30dff8a39","observation_id":"542d090a-bde5-476e-93c8-631aaba47c51","resolution":{"observed_at":"2026-08-07T12:37:28.544301Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.08295","last_updated":"2024-04-16T12:52:47Z","snapshot_observed_at":"2026-08-03T03:29:01.959523Z","submitted_at":"2024-03-13T06:59:16Z","title":"Gemma: Open Models Based on Gemini Research and Technology","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.08295","snapshot_observed_at":"2026-08-07T12:37:21.246351Z","title":"Gemma: Open models based on gemini research and technology","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:21.246351Z"},"links":{"cited_paper":"/paper/2403.08295","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:15fac6a878b26462684c7b33a7985852f65cddfb87c9d65ec7a2ee3f009e897f","observation_id":"fb5be959-6296-4046-9e8b-69de40a997a2","resolution":{"observed_at":"2026-08-07T12:37:21.246351Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:28.331323Z","title":"Cambrian-1: A fully open, vision- centric exploration of multimodal llms","venue":null,"work_id":"849743c3-f752-406f-95c5-955031387901","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:21.360353Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:84bfd5d030b0a615e3d69c13625badeb42a5436dd55a4fdb45fb5cba767cf522","observation_id":"0bb7053e-8653-4f74-a8ed-f522d94005a6","resolution":{"observed_at":"2026-08-07T12:37:28.389747Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-07T12:37:21.439956Z","title":"Llama 2: Open foundation and fine-tuned chat models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:21.439956Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:7cc36d2db7c5d439895c5c5043c54acaf6b26a817594b550da13b4804e2bcda9","observation_id":"a09a5b92-a3c3-46d7-9c28-29122d677e3c","resolution":{"observed_at":"2026-08-07T12:37:21.439956Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:28.133001Z","title":"Attention is all you need","venue":null,"work_id":"6a423123-6e96-44db-a556-c0e3ecb670e5","year":2017},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:21.559745Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:289cf2f47a8e3ec0c6963ddcbb7e241eec433070c84f4c86a3e44516ff54b725","observation_id":"46bcfb17-ae86-4fea-9e96-e026cddb3bfc","resolution":{"observed_at":"2026-08-07T12:37:28.221238Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:28.013766Z","title":"Show and tell: A neural image caption generator","venue":null,"work_id":"9eba579a-55b0-4803-84df-77af6f259ab6","year":2015},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:21.692309Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:63fe2f40457c40bea01bd11bfa58611d4d975a109be459879cc25b74a0b57b8e","observation_id":"a9736535-42eb-4e5f-8817-6603449fa2e5","resolution":{"observed_at":"2026-08-07T12:37:28.070943Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:27.820049Z","title":"Efficient large language models: A survey.Transactions on Machine Learning Research (TMLR), 2024","venue":null,"work_id":"45e0ece0-346a-4e4b-8036-d31040798588","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:21.768128Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:754ce2a908a924ba551016f21e9e9303bd5383dc5c2cb62791d93b8e0774c576","observation_id":"cf88f0d5-6501-4dae-b17b-c9deec7298aa","resolution":{"observed_at":"2026-08-07T12:37:27.896903Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:27.627446Z","title":"Weakly supervised gaussian contrastive grounding with large multimodal models for video question answering","venue":null,"work_id":"87438bdc-bd46-49d1-bfae-205cd545fc41","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:21.846512Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:2c1f0aab67c604c4f09dee300648ddcb6ec1eabb45aee2ad73f5e625f585487d","observation_id":"8e98f985-ca8e-4fc2-a759-67f3c756a9be","resolution":{"observed_at":"2026-08-07T12:37:27.715659Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12191","last_updated":"2024-10-03T15:54:49Z","snapshot_observed_at":"2026-08-06T05:35:29.109022Z","submitted_at":"2024-09-18T17:59:32Z","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12191","snapshot_observed_at":"2026-08-07T12:37:21.942911Z","title":"Qwen2-vl: Enhancing vision-language model’s perception of the world at any resolution","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:21.942911Z"},"links":{"cited_paper":"/paper/2409.12191","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:40eee836f4817bf082d266d315a322244255c83170e24f61212a621f3dcaed34","observation_id":"88731b3a-d85e-4308-ab9d-076a50bdcc2a","resolution":{"observed_at":"2026-08-07T12:37:21.942911Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.20504","last_updated":"2025-03-24T02:17:34Z","snapshot_observed_at":"2026-08-16T12:59:47.295631Z","submitted_at":"2024-12-29T15:42:24Z","title":"ReTaKe: Reducing Temporal and Knowledge Redundancy for Long Video Understanding","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.20504","snapshot_observed_at":"2026-08-07T12:37:22.035335Z","title":"Retake: Reducing temporal and knowledge redundancy for long video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:22.035335Z"},"links":{"cited_paper":"/paper/2412.20504","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:27003bc60a01cb76953fcfe4cc82a1dc28d0bcf373e110db5b8ec5859eefe415","observation_id":"26926a99-6fe9-4763-ae56-44f168a64c7d","resolution":{"observed_at":"2026-08-07T12:37:22.035335Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:27.433202Z","title":"Videoagent: Long-form video under- standing with large language model as agent","venue":null,"work_id":"d10c4782-c6f8-496e-9482-709c1412e708","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:22.139175Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:2344631a3e258cb3de73a4a4f790b28b1d607d4b03974f0afbf50647d0f66cdd","observation_id":"ae110e84-5baf-4ba8-94c9-2236d33e803d","resolution":{"observed_at":"2026-08-07T12:37:27.552815Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:27.274722Z","title":"Videotree: Adaptive tree-based video representation for llm reasoning on long videos","venue":null,"work_id":"81974ff7-ff60-478b-9d94-0d74cfe74a72","year":2025},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:22.236365Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:c2322a35fc54b0577d5c0a42088817d59a0dd5ea982ef27de033440651ace267","observation_id":"eb9742ac-5b10-4d64-a3ed-357f9bcc1d79","resolution":{"observed_at":"2026-08-07T12:37:27.342292Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:27.081580Z","title":"Dibs: Enhancing dense video captioning with unlabeled videos via pseudo boundary enrichment and online refinement","venue":null,"work_id":"60b64a48-d0be-4dc4-9984-61a9ba12fb1d","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:22.338769Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:acd4d1b1e09e985a09e443f4e343aaac0562c2e5134df5cfb32c999ab1fb800f","observation_id":"7cd90885-7735-4b77-8764-6ba6759e772e","resolution":{"observed_at":"2026-08-07T12:37:27.171688Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:26.889861Z","title":"Longvideobench: A benchmark for long-context interleaved video-language understanding","venue":null,"work_id":"4205c048-0e6f-499e-8e06-4e5ec5204bb5","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:22.450389Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:10659fbb46749fd77344ec342e85aab0abea00c4fce209a1e1214aa1e859d0c0","observation_id":"061e4745-6e27-4634-a3ca-076def2997d1","resolution":{"observed_at":"2026-08-07T12:37:26.974968Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:26.744859Z","title":"Next-qa: Next phase of question-answering to explaining temporal actions","venue":null,"work_id":"258426d4-95a0-44d6-9929-2f937df0317f","year":2021},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:22.581586Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:530693d8e0698e69e48d2026739725de714abfdda2090bde419f7173124a0675","observation_id":"7998ee8f-53e2-406a-944f-7d9d6e446f8b","resolution":{"observed_at":"2026-08-07T12:37:26.807231Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:26.579139Z","title":"Can i trust your answer? visually grounded video question answering","venue":null,"work_id":"bfcfb6d7-6666-4ee0-b933-71eee565a387","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:22.691030Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:05933ae6df77a35e15687e1a114b2ce7a9ecb24fca84245c387e0b5524d35865","observation_id":"519ac6c8-ad9d-415d-bb49-44f01783acce","resolution":{"observed_at":"2026-08-07T12:37:26.664010Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:26.404911Z","title":"Effective long-context scaling of foundation models","venue":null,"work_id":"843f68be-94ac-4ab6-9cd1-aa3305a4ccd8","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:22.786503Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:4762a5b3bf21048997b2c8bca7648d5549865cf32e73318406bb102b47dac99c","observation_id":"c3b083f2-ee51-4cb1-a99d-b3b5c2ef56c0","resolution":{"observed_at":"2026-08-07T12:37:26.500839Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16994","last_updated":"2024-04-29T14:52:02Z","snapshot_observed_at":"2026-08-13T20:40:43.794560Z","submitted_at":"2024-04-25T19:29:55Z","title":"PLLaVA : Parameter-free LLaVA Extension from Images to Videos for Video Dense Captioning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16994","snapshot_observed_at":"2026-08-07T12:37:22.900514Z","title":"Pllava: Parameter-free llava extension from images to videos for video dense captioning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:22.900514Z"},"links":{"cited_paper":"/paper/2404.16994","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:f5511e3df2499e3b8efd6d31bdf5e68329b50c0708e874c8845dc677c9deaf7d","observation_id":"a990e067-485f-4cf9-852f-ee77af800c35","resolution":{"observed_at":"2026-08-07T12:37:22.900514Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15841","last_updated":"2024-09-15T05:00:18Z","snapshot_observed_at":"2026-08-16T13:32:02.889396Z","submitted_at":"2024-07-22T17:58:04Z","title":"SlowFast-LLaVA: A Strong Training-Free Baseline for Video Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15841","snapshot_observed_at":"2026-08-07T12:37:23.032653Z","title":"Slowfast-llava: A strong training-free baseline for video large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:23.032653Z"},"links":{"cited_paper":"/paper/2407.15841","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:c7605944db25f0b3227cd57a3acc56e146f84d3144f1bd35707b9e8f2d8dd956","observation_id":"cba55f33-db7f-408d-99fa-78317f5baaca","resolution":{"observed_at":"2026-08-07T12:37:23.032653Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:26.269083Z","title":"Zero-shot video question answering via frozen bidirectional language models","venue":null,"work_id":"f01c814e-969e-4c68-9d93-936d20f0eba2","year":2022},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:23.130555Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:5383b61043e7b81c8f4a3a99736d31a6d765e1887bf3ba6d76c25b4358c3fd0a","observation_id":"cb3d9dd3-06aa-45e8-bdb6-fdb75c4eda2e","resolution":{"observed_at":"2026-08-07T12:37:26.332155Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:26.080203Z","title":"Vid2seq: Large-scale pretraining of a visual language model for dense video captioning","venue":null,"work_id":"d889a329-efd6-40b1-be78-beaf9bf792f9","year":2023},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:23.230304Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:5fc5d535b044f085a27dc35ba3833d5bea87e90dee3aa88efb3e12bb8ed3b65a","observation_id":"51652350-edb5-4a59-8448-5a2ab454b129","resolution":{"observed_at":"2026-08-07T12:37:26.186839Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:25.990715Z","title":"Dense connector for mllms","venue":null,"work_id":"69b8bb1b-0fa2-4eae-be92-9d7bf57831ba","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:23.342782Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:dff13afa24ee9143a756ae192a3814bd18bcfb0784046f05f87908afb6d18292","observation_id":"9298836e-13e0-461f-ab6c-1abfb7ea459e","resolution":{"observed_at":"2026-08-07T12:37:26.019452Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09146","last_updated":"2025-09-02T09:52:40Z","snapshot_observed_at":"2026-08-16T22:28:31.759830Z","submitted_at":"2025-03-12T08:16:39Z","title":"Generative Frame Sampler for Long Video Understanding","version":2},"cited_work":{"arxiv_id":"2503.09146","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.09146","snapshot_observed_at":"2026-08-07T12:37:24.797285Z","title":"Generative Frame Sampler for Long Video Understanding","venue":"cs.CV","work_id":"89c807d2-4027-4c52-8920-70732296db0b","year":2025},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:23.440876Z"},"links":{"cited_paper":"/paper/2503.09146","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:8ea20ed19dc94355cde4c0b8acc5e243e1eaa117a1b3255d173c39501ba20879","observation_id":"0ef5db10-d2dc-4306-898f-c0a74ce8d362","resolution":{"observed_at":"2026-08-07T12:37:24.889986Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.01800","last_updated":"2024-08-03T15:02:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-03T15:02:21Z","title":"MiniCPM-V: A GPT-4V Level MLLM on Your Phone","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.01800","snapshot_observed_at":"2026-08-07T12:37:23.584263Z","title":"Minicpm-v: A gpt-4v level mllm on your phone","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:23.584263Z"},"links":{"cited_paper":"/paper/2408.01800","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:1847d71ae2c51063ebea9f620b1b4da7f44c97b961d76dfc1698272fb1656228","observation_id":"e9d53f51-3628-4ccd-8161-bafbb69f4d9a","resolution":{"observed_at":"2026-08-07T12:37:23.584263Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:25.851332Z","title":"Self-chained image-language model for video localization and question answering","venue":null,"work_id":"a7a78d6b-5164-43c5-99d3-c2a0fbd268d6","year":2023},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:23.714914Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:0674c15e21ec1e90ef2e29cd39b4d43a5f6feb241d9e9ef208eab34fb7d195b9","observation_id":"1388fada-bbd9-4c81-83ce-e5b474ed503f","resolution":{"observed_at":"2026-08-07T12:37:25.913672Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:25.680294Z","title":"Frame-voyager: Learning to query frames for video large language models","venue":null,"work_id":"37fd5d8f-7494-4b46-8402-35f8ec78655c","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:23.826563Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:46cee39d08ebc80b3b202861f245a52b5a9a520982243e0b31bc10b80d1433e5","observation_id":"a2d295cf-2c3e-4d3b-864f-226d08d397da","resolution":{"observed_at":"2026-08-07T12:37:25.779824Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:25.476452Z","title":"Sigmoid loss for language image pre-training","venue":null,"work_id":"203685c9-afbf-46c9-9579-21399d548e8f","year":2023},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:23.882899Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:5ffec58d7e55152f9af22fe009721936e660564d2efccd5b118b441353ed6908","observation_id":"49e98b79-f970-452f-9e47-f1468db5e4da","resolution":{"observed_at":"2026-08-07T12:37:25.554077Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:25.330415Z","title":"A simple llm framework for long-range video question-answering","venue":null,"work_id":"ae041f53-3641-4972-9788-87f6524d1117","year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:23.986712Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:a87713f4e174023fa5c9ef5d3789f2cfdc6ed76fa700ab82ec3e487f09cac814","observation_id":"842a74e2-9b22-453c-b188-65da3f3cc9e1","resolution":{"observed_at":"2026-08-07T12:37:25.380735Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.16852","last_updated":"2024-07-01T02:59:29Z","snapshot_observed_at":"2026-08-07T09:52:45.942315Z","submitted_at":"2024-06-24T17:58:06Z","title":"Long Context Transfer from Language to Vision","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.16852","snapshot_observed_at":"2026-08-07T12:37:24.093163Z","title":"Long context transfer from language to vision","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:24.093163Z"},"links":{"cited_paper":"/paper/2406.16852","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:d5718d7008eece9bdc21b38a5e6f1164a0f56b8f808b2d79ecba2a45fa83065c","observation_id":"d0f16f33-24b5-4af9-ad4e-4b8e07355c6d","resolution":{"observed_at":"2026-08-07T12:37:24.093163Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:37:24.190869Z","title":"Llava-next: A strong zero-shot video understanding model, April 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:24.190869Z"},"links":{"citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:86f3907be081982bc6e08a55a4aa89b672f6d4260113d89674dded6b8146a5f1","observation_id":"79b9db04-5fc5-48b4-aadf-697adeb8f07e","resolution":{"observed_at":"2026-08-07T12:37:24.190869Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02713","last_updated":"2025-08-01T16:40:14Z","snapshot_observed_at":"2026-08-02T12:24:31.329178Z","submitted_at":"2024-10-03T17:36:49Z","title":"LLaVA-Video: Video Instruction Tuning With Synthetic Data","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02713","snapshot_observed_at":"2026-08-07T12:37:24.258273Z","title":"Video instruction tuning with synthetic data","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:24.258273Z"},"links":{"cited_paper":"/paper/2410.02713","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:6d444a8ce72518fa2f527b09af3800960cb0d754f9b429dafa5f2aca28d51513","observation_id":"b30ebedc-b77a-4463-b3a6-ef8c7a773f6d","resolution":{"observed_at":"2026-08-07T12:37:24.258273Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04264","last_updated":"2025-01-01T15:53:58Z","snapshot_observed_at":"2026-08-03T20:38:36.602554Z","submitted_at":"2024-06-06T17:09:32Z","title":"MLVU: Benchmarking Multi-task Long Video Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04264","snapshot_observed_at":"2026-08-07T12:37:24.363162Z","title":"Mlvu: A comprehensive benchmark for multi-task long video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:24.363162Z"},"links":{"cited_paper":"/paper/2406.04264","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:a504f19dd4d90ec8482453383866d90b00d797ad143ec94bd3fa4502f666edfb","observation_id":"b028f069-8365-4ff5-903f-0d9928d4f313","resolution":{"observed_at":"2026-08-07T12:37:24.363162Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.10592","last_updated":"2023-10-02T16:38:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-20T18:25:35Z","title":"MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.10592","snapshot_observed_at":"2026-08-07T12:37:24.461768Z","title":"Minigpt-4: Enhancing vision-language understanding with advanced large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:24.461768Z"},"links":{"cited_paper":"/paper/2304.10592","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:aa2cc8c4bede20b802153777ad3c5ac53122f458781f4b716f031639aeb59f61","observation_id":"4793def1-2e8a-4082-a776-2a2c5df55d59","resolution":{"observed_at":"2026-08-07T12:37:24.461768Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.10360","last_updated":"2024-12-13T18:53:24Z","snapshot_observed_at":"2026-08-15T17:40:10.309874Z","submitted_at":"2024-12-13T18:53:24Z","title":"Apollo: An Exploration of Video Understanding in Large Multimodal Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.10360","snapshot_observed_at":"2026-08-07T12:37:24.588694Z","title":"A. High heels","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-07T12:37:24.588694Z"},"links":{"cited_paper":"/paper/2412.10360","citing_paper":"/paper/2505.24158"},"observation_digest":"sha256:eead4469e06d77c869993e4971c5e2e7d317001f612e687aefae3886b05fb804","observation_id":"447bbede-b0be-4fbc-9feb-22827ba5a7bd","resolution":{"observed_at":"2026-08-07T12:37:24.588694Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2505.24158","last_updated":"2025-05-30T03:04:28Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-13T08:55:49.115006Z","submitted_at":"2025-05-30T03:04:28Z","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders"},"reference_resolution":{"displayed":84,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":30,"verified_exact":1,"verified_fuzzy":52},"total_outbound_references":84},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"thesis":"As of 16 August 2026, this Paper Citation Record lists 84 of 84 outbound references and 3 inbound Pith citation observations for arXiv:2505.24158."}