{"as_of":"2026-08-21T05:19:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:eebc8ec113536dae726239ae7bc63e82a85ec1a71a5a7dae361984ebb2059b6a","coverage":[{"denominator":46,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":46,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T10:43:17.664668Z","state":"measured"},{"denominator":53,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":53,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-20T06:33:59.587034+00:00","state":"measured"},{"denominator":7,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":7,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T14:57:42.167776Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T16:09:57.135194Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.17447","snapshot_observed_at":"2026-08-05T19:03:08.072560Z","title":"Frag: Frame selection augmented generation for long video and long document understanding","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.13692","last_updated":"2025-08-19T09:52:04Z","snapshot_observed_at":"2026-08-16T10:29:42.442699Z","submitted_at":"2025-08-19T09:52:04Z","title":"HumanPCR: Probing MLLM Capabilities in Diverse Human-Centric Scenes","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-05T19:03:08.072560Z"},"links":{"cited_paper":"/paper/2504.17447","citing_paper":"/paper/2508.13692"},"observation_digest":"sha256:41c729356a8ab1ea32ca3ddbc2ca1f0de73188844e20a0fdf93e57b7731c7950","observation_id":"34d43881-d5b2-4eb7-a79c-a3532b767e37","resolution":{"observed_at":"2026-08-05T19:03:08.072560Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.17447","snapshot_observed_at":"2026-08-04T13:27:56.481199Z","title":"Frag: Frame selec- tion augmented generation for long video and long document understanding.arXiv preprint arXiv:2504.17447,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2510.00705","last_updated":"2026-06-26T15:22:20Z","snapshot_observed_at":"2026-08-14T01:20:42.424375Z","submitted_at":"2025-10-01T09:20:51Z","title":"Training-free Uncertainty Guidance for Complex Visual Tasks with MLLMs","version":3},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-04T13:27:56.481199Z"},"links":{"cited_paper":"/paper/2504.17447","citing_paper":"/paper/2510.00705"},"observation_digest":"sha256:77f9105a671e56fa95b784dfd3574fbfa78b3121d7521f31b6eb761b56f009e5","observation_id":"b25ee4a3-3132-4bde-b546-d701ac7253df","resolution":{"observed_at":"2026-08-04T13:27:56.481199Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"cited_work":{"arxiv_id":"2504.17447","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.17447","snapshot_observed_at":"2026-07-04T16:09:57.135194Z","title":"Frag: Frame selection augmented generation for long video and long document understanding","venue":null,"work_id":"f48134fb-8bcc-42ad-a83d-5f17b0d437ea","year":2025},"citing_paper":{"arxiv_id":"2605.10762","last_updated":"2026-05-11T15:57:46Z","snapshot_observed_at":"2026-08-12T20:32:51.175035Z","submitted_at":"2026-05-11T15:57:46Z","title":"GridProbe: Posterior-Probing for Adaptive Test-Time Compute in Long-Video VLMs","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-12T03:46:31.117768Z"},"links":{"cited_paper":"/paper/2504.17447","citing_paper":"/paper/2605.10762"},"observation_digest":"sha256:b53e434cfc28f56479869f0a95b1ef41a9f2ebd4d6c19be6fc9404d4c6e48952","observation_id":"9928cae8-7e0d-41b4-9ec5-2f63929189fc","resolution":{"observed_at":"2026-05-12T06:56:31.830293Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"cited_work":{"arxiv_id":"2504.17447","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.17447","snapshot_observed_at":"2026-07-04T16:09:57.135194Z","title":"Frag: Frame selection augmented generation for long video and long document understanding","venue":null,"work_id":"f48134fb-8bcc-42ad-a83d-5f17b0d437ea","year":2025},"citing_paper":{"arxiv_id":"2606.24187","last_updated":"2026-06-24T04:06:43Z","snapshot_observed_at":"2026-08-12T20:32:05.568631Z","submitted_at":"2026-06-23T06:13:30Z","title":"Towards Fast and Effective Long Video Understanding of Multimodal Large Language Models via Adaptive Quasi-Gaussian Sampling","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-26T00:53:43.629684Z"},"links":{"cited_paper":"/paper/2504.17447","citing_paper":"/paper/2606.24187"},"observation_digest":"sha256:48354c88d0758f693ddb45a8b075237af402368f1776ef69d8f2b1736ea3f305","observation_id":"d481f219-c623-41d4-8a26-96118247b5cb","resolution":{"observed_at":"2026-07-04T16:09:57.137169Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"cited_work":{"arxiv_id":"2504.17447","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.17447","snapshot_observed_at":"2026-07-04T16:09:57.135194Z","title":"Frag: Frame selection augmented generation for long video and long document understanding","venue":null,"work_id":"f48134fb-8bcc-42ad-a83d-5f17b0d437ea","year":2025},"citing_paper":{"arxiv_id":"2607.00983","last_updated":"2026-07-01T14:19:24Z","snapshot_observed_at":"2026-08-13T08:15:37.802200Z","submitted_at":"2026-07-01T14:19:24Z","title":"QCA: Query- and Content-Aware Keyframe Selection for Long Video Understanding","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-07-02T14:09:54.549499Z"},"links":{"cited_paper":"/paper/2504.17447","citing_paper":"/paper/2607.00983"},"observation_digest":"sha256:22f58abae7482deb6945d222c86d0dd5b0fe43b4af1b1825db7fabdf9e49ccd9","observation_id":"4065af9c-3783-4808-9ff1-13758e62168b","resolution":{"observed_at":"2026-07-02T14:17:02.680402Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.17447","snapshot_observed_at":"2026-07-11T17:18:41.284513Z","title":"arXiv preprint arXiv:2504.17447 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.04559","last_updated":"2026-07-06T00:19:42Z","snapshot_observed_at":"2026-08-09T14:49:10.281004Z","submitted_at":"2026-07-06T00:19:42Z","title":"QSVideo: Query-Conditioned Semantic Temporal Retrieval for Video Understanding","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-07-11T17:18:41.284513Z"},"links":{"cited_paper":"/paper/2504.17447","citing_paper":"/paper/2607.04559"},"observation_digest":"sha256:53f87f3ecaad50b31a71f213413d0df30c46b230ef4b3de2a42943a080f23ac3","observation_id":"d598b017-5f0a-4bc8-9264-60f35e2e2fb8","resolution":{"observed_at":"2026-07-11T17:18:41.284513Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.17447","snapshot_observed_at":"2026-08-15T14:57:42.167776Z","title":"InProceedings of the IEEE/CVF conference on computer vision and pattern recognition, 24108–24118","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2608.03160","last_updated":"2026-08-18T12:21:35Z","snapshot_observed_at":"2026-08-21T05:11:11.134381Z","submitted_at":"2026-08-04T05:47:44Z","title":"Caved or Convinced: Temporal Sampling Gates Claim Deference in Video Large Language Models","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-15T14:57:42.167776Z"},"links":{"cited_paper":"/paper/2504.17447","citing_paper":"/paper/2608.03160"},"observation_digest":"sha256:8d78be3ac74f3ab69629a921532d9dcd708e5ff226c7afa226464fff3843f38c","observation_id":"917e71ba-b271-41a7-b130-9ebdc8cb9808","resolution":{"observed_at":"2026-08-15T14:57:42.167776Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2504.17447/citation-record","integrity":"/paper/2504.17447/integrity","json":"/paper/2504.17447/citation-record.json","paper":"/paper/2504.17447"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2308.12966","last_updated":"2023-10-13T02:41:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-24T17:59:17Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12966","snapshot_observed_at":"2026-08-16T10:43:17.451471Z","title":"Qwen-vl: A frontier large vision-language model with versatile abilities","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.451471Z"},"links":{"cited_paper":"/paper/2308.12966","citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:5a88c802025e117d9a0cc6e1d23f01ccafdfd80e6e0c125fe734a9f221f428a6","observation_id":"1ada4143-35d9-45db-94cb-e14e07ef3683","resolution":{"observed_at":"2026-08-16T10:43:17.451471Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05861","last_updated":"2024-05-31T15:22:58Z","snapshot_observed_at":"2026-08-20T01:15:01.756715Z","submitted_at":"2024-02-08T17:50:22Z","title":"Memory Consolidation Enables Long-Context Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05861","snapshot_observed_at":"2026-08-16T10:43:17.457003Z","title":"Mem- ory consolidation enables long-context video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.457003Z"},"links":{"cited_paper":"/paper/2402.05861","citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:a62f89c269e63fa707736dda9104d74f003b15826d3050eebad2be2b463e49e3","observation_id":"f89af446-a8ea-472b-903a-6491e0a3f9c1","resolution":{"observed_at":"2026-08-16T10:43:17.457003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:43:18.354603Z","title":"Scene text visual question answering","venue":null,"work_id":"b830dc28-aaf4-4777-9d90-8954c3a64762","year":2019},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.462083Z"},"links":{"citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:f45b472dc643820ef8b09dfee01ec92badd7d934798c969d59664f8ad7aecb0c","observation_id":"3b633a9e-24a4-4eea-b8d9-6ed4b62d17bd","resolution":{"observed_at":"2026-08-16T10:43:18.359169Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:43:18.340285Z","title":"Gram: Global reasoning for multi-page vqa","venue":null,"work_id":"ac66dd92-405e-4d85-826a-cb9acc522d04","year":2024},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.466985Z"},"links":{"citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:9ad34d16bb57044732a2236b11d72e13d25d4a95b539db25e63b247df572bb05","observation_id":"e6059f84-e991-4470-9dc6-8990444182dd","resolution":{"observed_at":"2026-08-16T10:43:18.344862Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16821","last_updated":"2024-04-29T20:24:30Z","snapshot_observed_at":"2026-08-17T14:16:52.244007Z","submitted_at":"2024-04-25T17:59:19Z","title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16821","snapshot_observed_at":"2026-08-16T10:43:17.471691Z","title":"How far are we to gpt-4v? closing the gap to commercial multimodal models with open-source suites","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.471691Z"},"links":{"cited_paper":"/paper/2404.16821","citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:fbba53c39a85564a790bbacbacd69b0065a38cc85b190bd75c073f96c1c0182e","observation_id":"178e8cc1-86ac-4fb2-8dca-3d66c9b14994","resolution":{"observed_at":"2026-08-16T10:43:17.471691Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07476","last_updated":"2024-10-30T06:49:54Z","snapshot_observed_at":"2026-08-14T16:25:22.654846Z","submitted_at":"2024-06-11T17:22:23Z","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.07476","snapshot_observed_at":"2026-08-16T10:43:17.476787Z","title":"Videollama 2: Advancing spatial- temporal modeling and audio understanding in video-llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.476787Z"},"links":{"cited_paper":"/paper/2406.07476","citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:b2a48f641a82a0fc768d8d2b894d9fb0983be2c2e7818007cfad54bc7cb1bb5a","observation_id":"ce5fd533-6a47-45dd-95a7-9eeefb8dfe10","resolution":{"observed_at":"2026-08-16T10:43:17.476787Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.06512","last_updated":"2024-04-09T17:59:32Z","snapshot_observed_at":"2026-08-18T00:33:47.386234Z","submitted_at":"2024-04-09T17:59:32Z","title":"InternLM-XComposer2-4KHD: A Pioneering Large Vision-Language Model Handling Resolutions from 336 Pixels to 4K HD","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.06512","snapshot_observed_at":"2026-08-16T10:43:17.482676Z","title":"Internlm-xcomposer2- 4khd: A pioneering large vision-language model handling resolutions from 336 pixels to 4k hd","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.482676Z"},"links":{"cited_paper":"/paper/2404.06512","citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:10b179eadafc6b97378601ffcd61f95f02589f5a680bf71914648cf00b959a1d","observation_id":"c822df91-9e47-4ab0-ab21-1567456bcd32","resolution":{"observed_at":"2026-08-16T10:43:17.482676Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.21075","last_updated":"2025-05-30T13:08:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-05-31T17:59:47Z","title":"Video-MME: The First-Ever Comprehensive Evaluation Benchmark of Multi-modal LLMs in Video Analysis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.21075","snapshot_observed_at":"2026-08-16T10:43:17.487971Z","title":"Video-mme: The first-ever compre- hensive evaluation benchmark of multi-modal llms in video analysis","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.487971Z"},"links":{"cited_paper":"/paper/2405.21075","citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:ae9ba92b9b88f6d3f129758e968ce7804945b5fdb0f2b659d59fbe758acaf1c7","observation_id":"6de5e810-6df3-4d82-b3ea-bae87cf764c5","resolution":{"observed_at":"2026-08-16T10:43:17.487971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:43:18.326144Z","title":"Ma-lmm: Memory-augmented large multimodal model for long-term video understanding","venue":null,"work_id":"7d75c4d3-d8d1-467b-b445-b0524e5290d9","year":2024},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.493137Z"},"links":{"citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:74c9c7b89afdcd0357bf65887d9a7cc034f18e35ec78f53b7450330595fc7f1b","observation_id":"af9e289b-df83-4efe-a58f-38d089121ef5","resolution":{"observed_at":"2026-08-16T10:43:18.331002Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-16T14:08:19.332089Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-08-16T10:43:17.497736Z","title":"mplug-docowl 1.5: Unified structure learning for ocr-free document understanding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.497736Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:09edf86606610061e96a0ea1c92d9aaada3f2c7205be9f001aebf8f23028fcbb","observation_id":"20baadda-17aa-49cc-bb4e-72a54323a851","resolution":{"observed_at":"2026-08-16T10:43:17.497736Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.03420","last_updated":"2024-09-09T05:36:27Z","snapshot_observed_at":"2026-08-16T13:21:07.303548Z","submitted_at":"2024-09-05T11:09:00Z","title":"mPLUG-DocOwl2: High-resolution Compressing for OCR-free Multi-page Document Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.03420","snapshot_observed_at":"2026-08-16T10:43:17.502876Z","title":"mplug-docowl2: High-resolution compressing for ocr- free multi-page document understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.502876Z"},"links":{"cited_paper":"/paper/2409.03420","citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:924bb3f3feda10af821f46f00fb5cf5779387387acf0a98a766d95ac97c2b3e2","observation_id":"e97737cd-0f35-4ae0-80d8-acfea68998ad","resolution":{"observed_at":"2026-08-16T10:43:17.502876Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.01744","last_updated":"2025-06-06T17:53:30Z","snapshot_observed_at":"2026-08-16T13:13:16.474767Z","submitted_at":"2024-10-02T16:55:01Z","title":"Leopard: A Vision Language Model For Text-Rich Multi-Image Tasks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.01744","snapshot_observed_at":"2026-08-16T10:43:17.507757Z","title":"Leopard: A vision language model for text-rich multi-image tasks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.507757Z"},"links":{"cited_paper":"/paper/2410.01744","citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:395bc6746719b85ec373b87c47b5759c9acd502eea95929d85cb4532d012e245","observation_id":"3c9a97cb-fba8-4e27-a9d7-e688c47c9bb3","resolution":{"observed_at":"2026-08-16T10:43:17.507757Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:43:18.312242Z","title":"Scsampler: Sampling salient clips from video for efficient action recog- nition","venue":null,"work_id":"0d1c5f8e-111f-4588-b382-77be0caa2e77","year":2019},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.513237Z"},"links":{"citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:b886eec3525d573fbf94b7c5ebc273e43e1af4440e103498f2e7f73b3f1ac2d6","observation_id":"5f6767b8-faa8-49bf-b35d-290f5056b60a","resolution":{"observed_at":"2026-08-16T10:43:18.316744Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11897","last_updated":"2024-08-19T12:47:11Z","snapshot_observed_at":"2026-08-16T14:33:30.767806Z","submitted_at":"2023-12-19T06:42:47Z","title":"Text-Conditioned Resampler For Long Form Video Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11897","snapshot_observed_at":"2026-08-16T10:43:17.518068Z","title":"Text-conditioned resam- pler for long form video understanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.518068Z"},"links":{"cited_paper":"/paper/2312.11897","citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:9865da99b96f6a38d19b8c66615e1af751dc889719eab48a7b877c04a0e1ee5f","observation_id":"b86dcd42-eb52-46b0-8cce-528cb082ef19","resolution":{"observed_at":"2026-08-16T10:43:17.518068Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:43:18.298520Z","title":"Building and better understanding vision- language models: insights and future directions., 2024","venue":null,"work_id":"b968560f-c9e6-4de2-9a13-c2791111f755","year":2024},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.522825Z"},"links":{"citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:4a7e2b17346fddd858e52d60cf4645b44aa3757ddd0e1e4cedfd674d5737a22b","observation_id":"36dc74f4-5747-4cbc-a24f-9190fe7a50ef","resolution":{"observed_at":"2026-08-16T10:43:18.303019Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03326","last_updated":"2024-10-26T16:35:13Z","snapshot_observed_at":"2026-08-18T11:56:50.710310Z","submitted_at":"2024-08-06T17:59:44Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03326","snapshot_observed_at":"2026-08-16T10:43:17.527084Z","title":"Llava-onevision: Easy visual task transfer","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.527084Z"},"links":{"cited_paper":"/paper/2408.03326","citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:bd05d6130a7218aa259134550143aeca8e9a41ff1df9471102724bdc6f0fb524","observation_id":"58863500-c589-487f-90ce-db3d2db49149","resolution":{"observed_at":"2026-08-16T10:43:17.527084Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.07895","last_updated":"2024-07-28T19:58:08Z","snapshot_observed_at":"2026-08-13T00:09:23.835117Z","submitted_at":"2024-07-10T17:59:43Z","title":"LLaVA-NeXT-Interleave: Tackling Multi-image, Video, and 3D in Large Multimodal Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.07895","snapshot_observed_at":"2026-08-16T10:43:17.532350Z","title":"Llava-next-interleave: Tackling multi-image, video, and 3d in large multimodal models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.532350Z"},"links":{"cited_paper":"/paper/2407.07895","citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:03dc25eafac7bd2457600bddaeb1cf5ea9864fdcca5ffcfebff9a9c250bccdf3","observation_id":"011b615d-ccd1-4494-8f3d-1ca725d6d4cb","resolution":{"observed_at":"2026-08-16T10:43:17.532350Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:43:17.536688Z","title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.536688Z"},"links":{"citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:03c436e4c0b4f0cf6fb36e74e5397213d902c0fb7864c7305c8506d8aa0b3b45","observation_id":"c1217fc4-556b-436b-9b8a-84ed79f83219","resolution":{"observed_at":"2026-08-16T10:43:17.536688Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:43:18.273942Z","title":"Vila: On pre-training for visual language models","venue":null,"work_id":"57c08a80-ac04-4ba2-bbb4-2b7b30555354","year":2024},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.541436Z"},"links":{"citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:82bcbe750eadfb460c7ec7d59dff462ad924baaad5504473495700dafb78e2e4","observation_id":"43381703-c98e-4d46-b49d-f46925d5cb91","resolution":{"observed_at":"2026-08-16T10:43:18.279282Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:43:17.546188Z","title":"Visual instruction tuning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.546188Z"},"links":{"citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:3b26b9e58efb94ca03f9bfc1f1aba983f85ad79235fc14c44f938e5bbd8567d4","observation_id":"bb7b0b27-0a16-4ee6-9ad6-9cfce0a83da2","resolution":{"observed_at":"2026-08-16T10:43:17.546188Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.01523","last_updated":"2024-11-12T04:37:44Z","snapshot_observed_at":"2026-08-16T17:16:07.034094Z","submitted_at":"2024-07-01T17:59:26Z","title":"MMLongBench-Doc: Benchmarking Long-context Document Understanding with Visualizations","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.01523","snapshot_observed_at":"2026-08-16T10:43:17.550648Z","title":"Mmlongbench-doc: Benchmarking long-context doc- ument understanding with visualizations","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.550648Z"},"links":{"cited_paper":"/paper/2407.01523","citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:ffaaa55f9405fb019e71c8566c0e8620986b4b8bd4bab67647c5c77be1e99d67","observation_id":"4759bcae-9a95-4113-ac08-8a535f435ccb","resolution":{"observed_at":"2026-08-16T10:43:17.550648Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:43:18.250228Z","title":"Egoschema: A diagnostic benchmark for very long- form video language understanding","venue":null,"work_id":"cbe80993-daa5-4a41-9e2c-7de621173b01","year":2023},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.555175Z"},"links":{"citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:b5ac974fb4b939a0d69f16972cd3cc4526f104cf2dc66c80c3afc52d71be950b","observation_id":"402462e8-cc6e-4395-b2f9-27734b8d7c44","resolution":{"observed_at":"2026-08-16T10:43:18.254615Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:43:17.559536Z","title":"Docvqa: A dataset for vqa on document images","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.559536Z"},"links":{"citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:29f714581c7149cec5b0eb3d1ca256f7dfc9e0d3f3a5f47441cc7884bdcb2945","observation_id":"4c7c1924-1e88-4e5e-b8fd-eb3bbd8c7894","resolution":{"observed_at":"2026-08-16T10:43:17.559536Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:43:17.564016Z","title":"Too many frames, not all useful: Efficient strategies for long- form video qa","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.564016Z"},"links":{"citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:0357fe718e0f8fd409ad067a96b4321c76c5d73a110fea14fb7efd6941d429b0","observation_id":"39c4baba-695d-4bbe-b9c2-6f2680e3f6f3","resolution":{"observed_at":"2026-08-16T10:43:17.564016Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.16009","last_updated":"2024-05-25T02:22:09Z","snapshot_observed_at":"2026-08-16T13:49:34.484002Z","submitted_at":"2024-05-25T02:22:09Z","title":"Streaming Long Video Understanding with Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.16009","snapshot_observed_at":"2026-08-16T10:43:17.567967Z","title":"Streaming long video understanding with large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.567967Z"},"links":{"cited_paper":"/paper/2405.16009","citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:557b91504b0688c3232cbb21d62facbcfc672053c136dc87b1d51db713960ae8","observation_id":"82fedbd9-d060-4a2a-be62-bceff9ec89ba","resolution":{"observed_at":"2026-08-16T10:43:17.567967Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.17434","last_updated":"2024-10-22T21:21:37Z","snapshot_observed_at":"2026-08-13T11:36:11.821356Z","submitted_at":"2024-10-22T21:21:37Z","title":"LongVU: Spatiotemporal Adaptive Compression for Long Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.17434","snapshot_observed_at":"2026-08-16T10:43:17.572620Z","title":"Kim, Bilge Soran, Raghuraman Krishnamoor- thi, Mohamed Elhoseiny, and Vikas Chandra","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.572620Z"},"links":{"cited_paper":"/paper/2410.17434","citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:467ed68ab79713fc672e43e47162dfb7aa67298a0e49802ad9554319432e893b","observation_id":"f1eed15c-fea4-476d-bc42-759301067f22","resolution":{"observed_at":"2026-08-16T10:43:17.572620Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:43:18.224447Z","title":"Slidevqa: A dataset for document visual question answering on multiple images","venue":null,"work_id":"77c29317-f06c-4d04-9462-1e5fc969c24b","year":2023},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.577304Z"},"links":{"citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:188b5a92d3a9fd2533632ff0dc7cbfbfa39c60951510adfac1f9b06048404ef4","observation_id":"1857db82-5077-4b20-9c24-09374eac1b50","resolution":{"observed_at":"2026-08-16T10:43:18.230330Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:43:18.209429Z","title":"Instructdoc: A dataset for zero-shot gener- alization of visual document understanding with instructions","venue":null,"work_id":"948f7ecb-dccd-4353-b968-a3b8b7f58114","year":2024},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.581262Z"},"links":{"citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:cffef57a9f792575ff064a6ce898bd7b482326fe0f00b79c5f818fee79346e61","observation_id":"d7b5f9fb-345a-4dae-acf8-1fe0e52fb72a","resolution":{"observed_at":"2026-08-16T10:43:18.214048Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:43:18.194807Z","title":"Hi- erarchical multimodal transformers for multipage docvqa","venue":null,"work_id":"38c68230-5e57-44ab-bba7-c7cae162da3d","year":2023},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.585932Z"},"links":{"citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:67c28a47fc9b0da85cc4c775ff64d340003f993a90acf92039306f7c6d6ab800","observation_id":"dfc16086-dddb-4efb-ad8e-79989b8bfdc0","resolution":{"observed_at":"2026-08-16T10:43:18.199451Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00634","last_updated":"2024-09-24T04:41:08Z","snapshot_observed_at":"2026-08-16T13:38:25.179414Z","submitted_at":"2024-06-30T09:21:01Z","title":"Tarsier: Recipes for Training and Evaluating Large Video Description Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.00634","snapshot_observed_at":"2026-08-16T10:43:17.590383Z","title":"Tarsier: Recipes for training and evaluating large video description models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.590383Z"},"links":{"cited_paper":"/paper/2407.00634","citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:46c61c41b4088966e25fda237bc3caef509177e66c63eee3afa4a7a6d28de926","observation_id":"1298d1a8-b6fb-4a92-8886-b831c7296f81","resolution":{"observed_at":"2026-08-16T10:43:17.590383Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:43:18.180354Z","title":"Lvbench: An extreme long video under- standing benchmark, 2024","venue":null,"work_id":"30e04f49-95b5-43f5-9a2f-4b296cdfff9a","year":2024},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.595458Z"},"links":{"citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:e19a64ed2f4dd90f7810973b197404200130ca46ec4b33024946f411e0f94695","observation_id":"58ae6ca0-c904-4f4e-ab8f-d3c7e6208b1d","resolution":{"observed_at":"2026-08-16T10:43:18.184832Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.10517","last_updated":"2024-03-15T17:57:52Z","snapshot_observed_at":"2026-08-16T14:09:21.368307Z","submitted_at":"2024-03-15T17:57:52Z","title":"VideoAgent: Long-form Video Understanding with Large Language Model as Agent","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.10517","snapshot_observed_at":"2026-08-16T10:43:17.599746Z","title":"Videoagent: Long-form video understand- ing with large language model as agent","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.599746Z"},"links":{"cited_paper":"/paper/2403.10517","citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:db7f5f26db388d18e1638756e68da37c451f4b12f159a56aff4dce28e39bd928","observation_id":"5206eac9-1b0d-441b-943d-4ca90021443f","resolution":{"observed_at":"2026-08-16T10:43:17.599746Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.19209","last_updated":"2025-03-14T13:57:16Z","snapshot_observed_at":"2026-08-17T21:35:27.995630Z","submitted_at":"2024-05-29T15:49:09Z","title":"VideoTree: Adaptive Tree-based Video Representation for LLM Reasoning on Long Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.19209","snapshot_observed_at":"2026-08-16T10:43:17.604526Z","title":"Videotree: Adaptive tree-based video representation for llm reasoning on long videos","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.604526Z"},"links":{"cited_paper":"/paper/2405.19209","citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:c7adbe1ad2ca2975bf50c9f8060a7fdcd2ad1d22b1ad7ac94a8f8bf4ef7ed566","observation_id":"a83f2be9-a27a-4baf-acab-f113d7f31e9a","resolution":{"observed_at":"2026-08-16T10:43:17.604526Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03384","last_updated":"2024-07-20T07:17:00Z","snapshot_observed_at":"2026-08-21T02:55:32.642026Z","submitted_at":"2024-04-04T11:33:29Z","title":"LongVLM: Efficient Long Video Understanding via Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.03384","snapshot_observed_at":"2026-08-16T10:43:17.609201Z","title":"Longvlm: Efficient long video un- derstanding via large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.609201Z"},"links":{"cited_paper":"/paper/2404.03384","citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:324acd0518643a63c93f9773fac0c85faaddb4c5618eff6e403c3fba63462a15","observation_id":"9b4584ac-7149-4d2c-8a41-910b8a38eea8","resolution":{"observed_at":"2026-08-16T10:43:17.609201Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15754","last_updated":"2024-07-22T16:00:55Z","snapshot_observed_at":"2026-08-15T01:28:11.776642Z","submitted_at":"2024-07-22T16:00:55Z","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15754","snapshot_observed_at":"2026-08-16T10:43:17.613816Z","title":"Longvideobench: A benchmark for long-context inter- leaved video-language understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.613816Z"},"links":{"cited_paper":"/paper/2407.15754","citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:331f5421ad5e6473ccf0cbe2beb47d747c78edf288ecaaa554c3c93c1dd47dc2","observation_id":"83dd7ddf-af62-4574-84b1-80f13ceefb27","resolution":{"observed_at":"2026-08-16T10:43:17.613816Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.13766","last_updated":"2025-03-11T17:31:27Z","snapshot_observed_at":"2026-08-18T11:01:07.019808Z","submitted_at":"2024-07-18T17:59:30Z","title":"Visual Haystacks: A Vision-Centric Needle-In-A-Haystack Benchmark","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.13766","snapshot_observed_at":"2026-08-16T10:43:17.618225Z","title":"Visual haystacks: A vision-centric needle-in-a- haystack benchmark","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.618225Z"},"links":{"cited_paper":"/paper/2407.13766","citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:f6dcb2558131588526ba56a2e3faba31c46fc8dbe2692532a1608d5a1258fd25","observation_id":"354fad58-8c66-4189-8728-22959c02b8fa","resolution":{"observed_at":"2026-08-16T10:43:17.618225Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:43:18.166214Z","title":"Next-qa: Next phase of question-answering to explaining temporal actions","venue":null,"work_id":"3fae33be-2d0f-4b0b-a7a0-f7f2ac7a0620","year":2021},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.623636Z"},"links":{"citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:4c0be0e55fa2097073ad5edf4565bfcde33bb035eb2161cdb4c5cc4f7b625d05","observation_id":"ef481088-fd65-43c7-9e6e-30081001d4aa","resolution":{"observed_at":"2026-08-16T10:43:18.170973Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-08-14T12:07:36.260067Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.05970","snapshot_observed_at":"2026-08-16T10:43:17.627836Z","title":"Wukong: A large multimodal model for efficient long pdf reading with end-to-end sparse sampling","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.627836Z"},"links":{"cited_paper":"/paper/2410.05970","citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:6feda61aac8038514752e543601e7ff925b0108643b031651f04672ae51c8e0f","observation_id":"8c82477e-1992-47de-bbd0-7070ecd3c9bb","resolution":{"observed_at":"2026-08-16T10:43:17.627836Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.10188","last_updated":"2024-12-13T02:32:06Z","snapshot_observed_at":"2026-08-16T04:27:36.181491Z","submitted_at":"2024-08-19T17:48:08Z","title":"LongVILA: Scaling Long-Context Visual Language Models for Long Videos","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.10188","snapshot_observed_at":"2026-08-16T10:43:17.633250Z","title":"Longvila: Scaling long-context visual language models for long videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.633250Z"},"links":{"cited_paper":"/paper/2408.10188","citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:50bb8aff4a515bed28cac23cdf4434e7ebd2041f7b47960d2547aabb22eb4ad4","observation_id":"5d5ac8eb-d889-43ed-a7b2-f3af9282befb","resolution":{"observed_at":"2026-08-16T10:43:17.633250Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:43:18.150748Z","title":"Self-chained image-language model for video localization and question answering","venue":null,"work_id":"b9cb86ca-9913-448f-9d67-7ca5738b2166","year":2024},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.637791Z"},"links":{"citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:6de6c0bb9d0be81e3306d669fe99205f2b0a14ace7f3947fad09e8237ec74b0f","observation_id":"c4f6d4a8-51e4-4acc-9cb6-8c8208e0830f","resolution":{"observed_at":"2026-08-16T10:43:18.156588Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03226","last_updated":"2025-03-28T03:19:52Z","snapshot_observed_at":"2026-08-20T09:36:35.686768Z","submitted_at":"2024-10-04T08:26:06Z","title":"Frame-Voyager: Learning to Query Frames for Video Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.03226","snapshot_observed_at":"2026-08-16T10:43:17.642133Z","title":"Frame-voyager: Learn- ing to query frames for video large language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.642133Z"},"links":{"cited_paper":"/paper/2410.03226","citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:a5969cd40189b826719dd97f95ded4ce94b6c48d53b3da026992b98e2b477252","observation_id":"a1c67c44-1506-41e7-bf7e-73884a4d6389","resolution":{"observed_at":"2026-08-16T10:43:17.642133Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:43:17.646612Z","title":"Sigmoid loss for language image pre-training","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.646612Z"},"links":{"citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:c7d1ef950f07ac906986b5d697b14f2b42e742cf20c85ce992d8124b3b5248c2","observation_id":"3d8254d3-2911-476a-afff-4abea8b596ce","resolution":{"observed_at":"2026-08-16T10:43:17.646612Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.16852","last_updated":"2024-07-01T02:59:29Z","snapshot_observed_at":"2026-08-07T09:52:45.942315Z","submitted_at":"2024-06-24T17:58:06Z","title":"Long Context Transfer from Language to Vision","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.16852","snapshot_observed_at":"2026-08-16T10:43:17.650992Z","title":"Long context transfer from language to vision","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.650992Z"},"links":{"cited_paper":"/paper/2406.16852","citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:9bbdf876d4c41df4ab183956cf0660a83c81dc7459c2e1d66dbe0d6b08472f53","observation_id":"f8a7fa0d-12d6-43e3-bddb-2f8f684dfb1c","resolution":{"observed_at":"2026-08-16T10:43:17.650992Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:43:17.655560Z","title":"Llava- next: A strong zero-shot video understanding model, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.655560Z"},"links":{"citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:21e58d3d3383ff7739de8d133d7f22844a49a8c54ec97beabc36e588a7b311b9","observation_id":"5d3a52f5-c8a9-468c-855a-24bd4205849f","resolution":{"observed_at":"2026-08-16T10:43:17.655560Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04264","last_updated":"2025-01-01T15:53:58Z","snapshot_observed_at":"2026-08-03T20:38:36.602554Z","submitted_at":"2024-06-06T17:09:32Z","title":"MLVU: Benchmarking Multi-task Long Video Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04264","snapshot_observed_at":"2026-08-16T10:43:17.660147Z","title":"Mlvu: A comprehensive benchmark for multi-task long video understanding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.660147Z"},"links":{"cited_paper":"/paper/2406.04264","citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:4d94676958bc2ff91dee1a23e8ff876a9db5a8ba01f8c89fa492a22092557647","observation_id":"272592ad-4daa-4e37-95db-74214230670d","resolution":{"observed_at":"2026-08-16T10:43:17.660147Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:43:18.115765Z","title":"Selection Scoring Prompt Our prompt is based on the prompt used in SeViLA [40]","venue":null,"work_id":"d9501257-6ce1-4e55-86ad-75248e7e7981","year":null},"citing_paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-16T10:43:17.664668Z"},"links":{"citing_paper":"/paper/2504.17447"},"observation_digest":"sha256:71f6b814cad8ed3efc33adf7c7c13148c26650834b8153d833fe3c9311875c5d","observation_id":"f38d4079-ca5a-428e-a6af-ee1209482b68","resolution":{"observed_at":"2026-08-16T10:43:18.122194Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2504.17447","last_updated":"2025-04-24T11:19:18Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-20T12:51:37.984627Z","submitted_at":"2025-04-24T11:19:18Z","title":"FRAG: Frame Selection Augmented Generation for Long Video and Long Document Understanding"},"reference_resolution":{"displayed":46,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":32,"verified_exact":0,"verified_fuzzy":14},"total_outbound_references":46},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"thesis":"As of 21 August 2026, this Paper Citation Record lists 46 of 46 outbound references and 7 inbound Pith citation observations for arXiv:2504.17447."}