{"as_of":"2026-08-10T15:21:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:766b469565c6bb17f0f8ff2b63fa247cb536bad9f71d42465940338c72366ac6","coverage":[{"denominator":57,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":57,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T15:34:44.512500Z","state":"measured"},{"denominator":57,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":57,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2507.15569/citation-record","integrity":"/paper/2507.15569/integrity","json":"/paper/2507.15569/citation-record.json","paper":"/paper/2507.15569"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.354775Z","title":"Flamingo: a visual language model for few-shot learning.NeurIPS, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.354775Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:d2328cf4d813659a76c24f4b3089b65aaf5f4c4a552eb036247853efe913bdc6","observation_id":"f2aa759d-540a-4dd3-9b35-fcdbadc800c3","resolution":{"observed_at":"2026-08-06T15:34:44.354775Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.358353Z","title":"Vivit: A video vision transformer","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.358353Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:d67b60dbd51a080559722b543c7e5d5402520573685c98b4c50f710e9fa66e52","observation_id":"f12f1706-f49e-452d-bde1-3937039bfe13","resolution":{"observed_at":"2026-08-06T15:34:44.358353Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.17274","last_updated":"2022-06-03T17:52:04Z","snapshot_observed_at":"2026-08-10T12:13:29.624714Z","submitted_at":"2022-03-31T17:59:30Z","title":"Exploring Visual Prompts for Adapting Large-Scale Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.17274","snapshot_observed_at":"2026-08-06T15:34:44.361412Z","title":"Exploring visual prompts for adapting large- scale models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.361412Z"},"links":{"cited_paper":"/paper/2203.17274","citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:e6249b18155534bc10136c4a6b248c1b0dd69134e2a2528ddc7ed81e552a5d8c","observation_id":"a810cc3f-b520-4e8b-837c-051629199850","resolution":{"observed_at":"2026-08-06T15:34:44.361412Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:45.182988Z","title":"Relevant intrinsic feature enhancement network for few-shot semantic segmentation","venue":null,"work_id":"0dc3ed80-a0d8-4e59-90c4-39f53536db9d","year":2024},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.364929Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:50c7c46bae3b07de58386a53968600d21e70ff64b1a45f8b3f064d540c9494d6","observation_id":"9204a1de-db76-4350-988f-6541a262e604","resolution":{"observed_at":"2026-08-06T15:34:45.186177Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:45.173944Z","title":"Cores: Orchestrating the dance of reasoning and seg- mentation","venue":null,"work_id":"abb71ff2-73e1-4976-98b6-af5c2a387979","year":2024},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.368268Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:50ff09c75662e218b90c2ab2ea24e1f8f3bbf75d769806735222502db1974ec4","observation_id":"d902335b-31c9-4ce7-b95e-cd24b9314925","resolution":{"observed_at":"2026-08-06T15:34:45.177037Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.371708Z","title":"Is space-time attention all you need for video understanding? In ICML, page 4, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.371708Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:fa992908f4c3c2a67a5739ee9c2e77ffa637cb8ee47058b1683f2da78d9f7662","observation_id":"97a9d4ab-5612-42dc-9568-a85eafd9d064","resolution":{"observed_at":"2026-08-06T15:34:44.371708Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:45.158496Z","title":"Activitynet: A large-scale video benchmark for human activity understanding","venue":null,"work_id":"7a1e45b2-9cb6-494f-b4ae-7c6e030e9b05","year":2015},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.374741Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:4fb499af4357c1e9fc5d3d84a3a5b754a0671c1bfd8f826bff4586e680ca4864","observation_id":"ed27e943-8da2-48a9-b119-bd9c8fb48b69","resolution":{"observed_at":"2026-08-06T15:34:45.161693Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:45.148580Z","title":"Quo vadis, action recognition? a new model and the kinetics dataset","venue":null,"work_id":"5e0f7318-0226-4333-85b5-1a4762cbe712","year":2017},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.377884Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:7734d43cac4f6e48d0386e1e9364a645c630e5f597c4c4280d56408f26659182","observation_id":"047a7701-6c38-4280-b268-a02799f653c6","resolution":{"observed_at":"2026-08-06T15:34:45.151912Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:45.138965Z","title":"Deep temporal linear encoding networks","venue":null,"work_id":"07849ac5-5f1c-49ee-9a88-f914147cad22","year":2017},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.380704Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:b51ca398959377aec805bad50e992f4debb6053474da814b69ffa8e67cb88e0d","observation_id":"31550c44-a080-4788-aebf-c472fb184c73","resolution":{"observed_at":"2026-08-06T15:34:45.142530Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-08-10T01:12:16.468283Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-06T15:34:44.383416Z","title":"An image is worth 16x16 words: Trans- formers for image recognition at scale","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.383416Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:13f77be9eeee0d1cef8c9d0a352903ab2cd6daf187ffc7341685b4b64e5ef241","observation_id":"01a65432-7fe1-4d89-abb6-9290af5a52c7","resolution":{"observed_at":"2026-08-06T15:34:44.383416Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:45.128691Z","title":"Convolutional two-stream network fusion for video action recognition","venue":null,"work_id":"c7d6b4bf-8d52-4775-93bf-407a71a862a8","year":1933},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.386251Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:0513749733ece16a0492cc832deb09fa8d956ca7bd2400208e8b11e0d6b0b3ce","observation_id":"efca8d38-954e-475a-8066-dd34ff2a7bde","resolution":{"observed_at":"2026-08-06T15:34:45.131871Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.388931Z","title":"Slowfast networks for video recognition","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.388931Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:7bca6b6cad424199433ef13b377d9ca31b32f5bc538c7e64de760f7d734e6805","observation_id":"06b216d4-bf76-415d-bfec-5b4821cd3b8e","resolution":{"observed_at":"2026-08-06T15:34:44.388931Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.12980","last_updated":"2023-07-24T17:58:06Z","snapshot_observed_at":"2026-08-08T23:43:33.914655Z","submitted_at":"2023-07-24T17:58:06Z","title":"A Systematic Survey of Prompt Engineering on Vision-Language Foundation Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.12980","snapshot_observed_at":"2026-08-06T15:34:44.391857Z","title":"A systematic survey of prompt engineer- ing on vision-language foundation models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.391857Z"},"links":{"cited_paper":"/paper/2307.12980","citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:27abc3cf4bc348729bb3c54e2b5a708c767a52175b3f4f6f2d15bba5e5c68716","observation_id":"fe50a836-005b-4f99-9b63-5f8df3e74e80","resolution":{"observed_at":"2026-08-06T15:34:44.391857Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:45.112223Z","title":"Lita: Language instructed temporal-localization assistant","venue":null,"work_id":"7c1a8927-580f-4008-a8c3-a4ab4b08e2e0","year":2024},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.394727Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:e4265a70dec651c686a931c937eb1701380c56df3ef0d6fa8acd2479255cf87b","observation_id":"6c83cb25-e520-4df3-9bd8-c0d822c1f4a9","resolution":{"observed_at":"2026-08-06T15:34:45.115808Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.397334Z","title":"Vi- sual prompt tuning","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.397334Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:8146a7420873b9df4bb05139aa01607820d389382df9cf7dded592681c019625","observation_id":"8758f006-c3e4-4eaf-bf76-4b97f6f07d39","resolution":{"observed_at":"2026-08-06T15:34:44.397334Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:45.096056Z","title":"Chat-univi: Unified visual representation em- powers large language models with image and video un- derstanding","venue":null,"work_id":"ba58cba5-38da-47ed-aef5-07d9d83e8905","year":2024},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.400058Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:9850278fc4767b1c98795836d942e7a01a17d2df4255bae7fcc7b36a4243b5f4","observation_id":"76e16600-94a2-4da5-a431-b8651859cbbc","resolution":{"observed_at":"2026-08-06T15:34:45.099303Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.402876Z","title":"Video-lavit: Unified video-language pre-training with decoupled visual-motional tokenization","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.402876Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:44f67b0b9e7b5e4e7d3d931734be5a36e382db323d730f4e8ff44694993ae667","observation_id":"6b2d06ef-f871-479b-bf41-758ceadb3ae8","resolution":{"observed_at":"2026-08-06T15:34:44.402876Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:45.083889Z","title":"Large-scale video classification with convolutional neural networks","venue":null,"work_id":"6e572fc5-40c9-41f0-93e1-54d607932090","year":2014},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.405446Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:d3b90033e7c0b212e56f355a176f2baada829278ba1f70ff8b6b2b03995451e5","observation_id":"6efde5e5-e056-4173-8d90-cf61eefb4bd5","resolution":{"observed_at":"2026-08-06T15:34:45.089195Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:45.073978Z","title":"Maple: Multi-modal prompt learning","venue":null,"work_id":"cc9e019f-4432-4d9b-892a-49d424bd645f","year":2023},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.407980Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:937913006226a2ffbee281bea54ce5fc5ee9c49d3ab216ae8d6647ca1b9c376b","observation_id":"886d5255-bf2c-432e-9ee2-bafee1ce1371","resolution":{"observed_at":"2026-08-06T15:34:45.077119Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.18406","last_updated":"2024-03-27T09:48:23Z","snapshot_observed_at":"2026-07-06T17:51:48.378864Z","submitted_at":"2024-03-27T09:48:23Z","title":"An Image Grid Can Be Worth a Video: Zero-shot Video Question Answering Using a VLM","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.18406","snapshot_observed_at":"2026-08-06T15:34:44.410539Z","title":"An image grid can be worth a video: Zero- shot video question answering using a vlm","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.410539Z"},"links":{"cited_paper":"/paper/2403.18406","citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:997c0b8d62192009f2d93386927b92586b75714875054a4c04e86ba9ce64f477","observation_id":"86d87ae0-451a-468d-ba72-f19c6f13db1c","resolution":{"observed_at":"2026-08-06T15:34:44.410539Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.00692","last_updated":"2024-05-01T05:10:13Z","snapshot_observed_at":"2026-08-06T07:43:56.889679Z","submitted_at":"2023-08-01T17:50:17Z","title":"LISA: Reasoning Segmentation via Large Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.00692","snapshot_observed_at":"2026-08-06T15:34:44.413798Z","title":"Lisa: Reasoning segmentation via large language model","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.413798Z"},"links":{"cited_paper":"/paper/2308.00692","citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:813f42f816fa9b0cbade74603b1248b650a59bca0ff2cec7a20aa8b5cd094f73","observation_id":"78686686-890a-4340-87c9-c72b035b5a40","resolution":{"observed_at":"2026-08-06T15:34:44.413798Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:45.064498Z","title":"Deep local video feature for action recog- nition","venue":null,"work_id":"199c6a3f-e9e9-44b3-90dd-2bc7ed1e553e","year":2017},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.416913Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:2988fcba27f52fa528a90f5266d196b97d4a49d005e47adfff516224d53f0679","observation_id":"e67cff02-1234-4df9-85db-504bce9b20d6","resolution":{"observed_at":"2026-08-06T15:34:45.067593Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.12597","last_updated":"2023-06-15T07:57:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-01-30T00:56:51Z","title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.12597","snapshot_observed_at":"2026-08-06T15:34:44.419656Z","title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.419656Z"},"links":{"cited_paper":"/paper/2301.12597","citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:daefa916396c9fe1d0378d6f0bde73cc5e365231293689100de92a749724bb87","observation_id":"cbe3b580-50c7-4eb6-9bc5-41608f5d5610","resolution":{"observed_at":"2026-08-06T15:34:44.419656Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06355","last_updated":"2024-01-04T02:06:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-10T17:59:04Z","title":"VideoChat: Chat-Centric Video Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06355","snapshot_observed_at":"2026-08-06T15:34:44.422636Z","title":"Videochat: Chat-centric video understanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.422636Z"},"links":{"cited_paper":"/paper/2305.06355","citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:1a57a53743e46a2b2128a2bcbeb6fedfbefd32c554a4c8683cb7879c14a9282f","observation_id":"1fe9950b-56f2-4ab2-b4a8-f37eaedb4cb7","resolution":{"observed_at":"2026-08-06T15:34:44.422636Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:45.054273Z","title":"Mvbench: A comprehensive multi-modal video understand- ing benchmark","venue":null,"work_id":"0929c9d2-b2fe-4532-9709-9cbcb6a29602","year":2024},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.425250Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:341e07eb1bc0c2dc04d42979a60dc223e532b88ab49cc16a499e055f2ea85940","observation_id":"3906ca21-74a7-436b-9fba-b566579c0520","resolution":{"observed_at":"2026-08-06T15:34:45.057513Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:45.044761Z","title":"Tgif: A new dataset and benchmark on animated gif description","venue":null,"work_id":"d958fa0b-bbd1-4167-913a-4aa41f7f1c1d","year":2016},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.427983Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:134e8b435dd9c8fc158ad7ca45616b3058606afd43bd0569be933a65eed4a97a","observation_id":"4f78f31d-0889-4c27-ab81-189ccb23a05e","resolution":{"observed_at":"2026-08-06T15:34:45.048001Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:45.034753Z","title":"Llama-vid: An image is worth 2 tokens in large language models","venue":null,"work_id":"810798a9-d271-4e46-b726-22b352104683","year":2025},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.430570Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:888dbfd456461cf3725bc00d3ae7f49cefad58c0940c022b068c22f230410781","observation_id":"5d3997aa-1479-4679-a207-0baac91a52ba","resolution":{"observed_at":"2026-08-06T15:34:45.038073Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.10122","last_updated":"2024-10-01T12:07:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-11-16T10:59:44Z","title":"Video-LLaVA: Learning United Visual Representation by Alignment Before Projection","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.10122","snapshot_observed_at":"2026-08-06T15:34:44.433210Z","title":"Video-llava: Learning united visual rep- resentation by alignment before projection","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.433210Z"},"links":{"cited_paper":"/paper/2311.10122","citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:0e7bb566fe68ec11734f0745ecaf836370aa796ccd8497144c04715fe08ade06","observation_id":"927376e1-c046-4e7c-9352-a6564ff8a61d","resolution":{"observed_at":"2026-08-06T15:34:44.433210Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.436170Z","title":"Visual instruction tuning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.436170Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:4065441de649add7205fa0080631e4c01306e721f86755488f88b0c9e9367dad","observation_id":"42037cba-cf27-4151-b814-ad705284f57f","resolution":{"observed_at":"2026-08-06T15:34:44.436170Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:45.018683Z","title":"Pre-train, prompt, and predict: A systematic survey of prompting methods in nat- ural language processing","venue":null,"work_id":"9236db6b-9427-4c26-b154-d949392f8259","year":2023},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.438941Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:ef9d2594980571394636acffdf609907f940fdcd51895d1ccabab868e6843a6b","observation_id":"f149dbf1-c940-4306-ba9b-dcb387911ee1","resolution":{"observed_at":"2026-08-06T15:34:45.022210Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:45.008077Z","title":"St-llm: Large language models are effective tem- poral learners","venue":null,"work_id":"235bfd06-55f5-41d4-a966-3f0c582277b3","year":2024},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.441777Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:7f4b5b0bebc036d2b63b3c723e4f4363997ab51cbe809c1274ed5f56473f9929","observation_id":"f949c9ea-8c37-4d96-aab7-71637291e73b","resolution":{"observed_at":"2026-08-06T15:34:45.011156Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02327","last_updated":"2026-05-01T17:44:24Z","snapshot_observed_at":"2026-07-06T19:45:00.034364Z","submitted_at":"2024-11-04T17:50:36Z","title":"PPLLaVA: Varied Video Sequence Understanding With Prompt Guidance","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.02327","snapshot_observed_at":"2026-08-06T15:34:44.444553Z","title":"Ppllava: Varied video se- quence understanding with prompt guidance","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.444553Z"},"links":{"cited_paper":"/paper/2411.02327","citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:0ef973ccb3c6daade4f04c65fcff48d7226846da2c513d91ca051bbf1365239e","observation_id":"0a00f510-85b7-4864-9be1-0d8183088207","resolution":{"observed_at":"2026-08-06T15:34:44.444553Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.998729Z","title":"Hybrid-level instruction injection for video token com- pression in multi-modal large language models","venue":null,"work_id":"1672e6fd-957b-4e2a-a522-cb55c3fd81c9","year":2025},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.447387Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:966ce004cd0d1d998f7eb73c944ae11c7a7f9c28d343f47fba624d9a680d880f","observation_id":"25f1e70a-1b36-485c-9ad4-bcbe0a679412","resolution":{"observed_at":"2026-08-06T15:34:45.001920Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.08870","last_updated":"2025-03-03T17:58:12Z","snapshot_observed_at":"2026-08-10T13:01:49.384599Z","submitted_at":"2023-12-12T09:47:59Z","title":"Vista-LLaMA: Reducing Hallucination in Video Language Models via Equal Distance to Visual Tokens","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.08870","snapshot_observed_at":"2026-08-06T15:34:44.450026Z","title":"Vista-llama: Reliable video narrator via equal distance to visual tokens","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.450026Z"},"links":{"cited_paper":"/paper/2312.08870","citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:3481a6cc299209d0a41410f5384f474595a29d9a8c4fe7e87cba8b67f1c9f5cb","observation_id":"22f7bdd3-7318-42a1-9806-3829e0bcfd82","resolution":{"observed_at":"2026-08-06T15:34:44.450026Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.989363Z","title":"Video-chatgpt: Towards detailed video understanding via large vision and language models","venue":null,"work_id":"d9218b6f-0ba5-46ee-a016-a1dbd887866a","year":2024},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.453154Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:d6ae68405c65cd5c0d86729d9b5606806725366a8c2d2557dcf91e7a0d280889","observation_id":"b074dd2b-5d0c-4053-9b0b-bcacb9f59593","resolution":{"observed_at":"2026-08-06T15:34:44.992531Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.979033Z","title":"Learning transferable visual models from natural language supervi- sion","venue":null,"work_id":"ffa44f55-77e4-4413-b576-b8e1a8ef9d59","year":2021},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.456204Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:9e4d8e440b4384c816065da4d230de903b6b42221224b48bbf6760a7475febcf","observation_id":"a26b246b-3b7a-4459-9e97-76a92169cba1","resolution":{"observed_at":"2026-08-06T15:34:44.982375Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.968760Z","title":"Mul- titask vision-language prompt tuning","venue":null,"work_id":"b1182a12-8b9e-44e5-a19a-1811b096a8f1","year":2024},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.458932Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:1bbccca793600c0ab014e6b02dbf959b1b0c956a2294f96aefa586cd0610954e","observation_id":"0b4c4c9b-ef73-4f35-a271-363c8546f5fe","resolution":{"observed_at":"2026-08-06T15:34:44.971774Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.958708Z","title":"What does clip know about a red circle? vi- sual prompt engineering for vlms","venue":null,"work_id":"01c37613-e36e-476a-b3ba-8763609c0a94","year":2023},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.461665Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:0cf32f19f04e3097928bcfbc328cbe475cbc9dde111014a14166f86ad54cad9d","observation_id":"a007cafb-bd39-468d-850d-151dc2858a56","resolution":{"observed_at":"2026-08-06T15:34:44.962147Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.949553Z","title":"Two-stream con- volutional networks for action recognition in videos","venue":null,"work_id":"965513ab-01ea-4bfd-b865-aa2587559fd3","year":2014},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.464208Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:2a71a696c1b5a9b5e4019d10dcb0197fd498d4ea7bdb499cd3044eafa1d1e290","observation_id":"0b9390b3-047e-40e2-885b-53296c9df236","resolution":{"observed_at":"2026-08-06T15:34:44.952584Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.939561Z","title":"Moviechat: From dense token to sparse memory for long video understanding","venue":null,"work_id":"9e562606-8f18-46be-8233-1fd93cc030bc","year":2024},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.466804Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:c6721351f69dc453339397506c19b14660c6bdf3bef0d0b6fa197642110ec5f9","observation_id":"8392e049-9274-45a9-a852-e6fb520459f0","resolution":{"observed_at":"2026-08-06T15:34:44.943100Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.469414Z","title":"Ufo: A unified approach to fine-grained visual perception via open- ended language interface","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.469414Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:0bfdc4e3331324d16f54182c60be234eaf6f9bae419602398d7cd9545131dfe3","observation_id":"5339d259-49d9-44ea-833b-97d2c51089c3","resolution":{"observed_at":"2026-08-06T15:34:44.469414Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.929911Z","title":"Qwen2.5: A party of foundation models, 2024","venue":null,"work_id":"a942dd9f-5fe7-44d3-a182-40237d879652","year":2024},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.472057Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:a04e444816d41c04d884dd8d312017be4a5642172b8867efd44041b639d27382","observation_id":"f116e752-67e6-49a2-b634-1073b17f5477","resolution":{"observed_at":"2026-08-06T15:34:44.932971Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.474572Z","title":"Learning spatiotemporal features with 3d convolutional networks","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.474572Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:68bf6caccfe9291ebe4a5a3844ae70b1eb2426383c244d667176f8b4fdac1b10","observation_id":"90ac6e1f-5dd4-4f91-a9d4-2e4a9e11e204","resolution":{"observed_at":"2026-08-06T15:34:44.474572Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.913882Z","title":"A closer look at spatiotemporal convolutions for action recognition","venue":null,"work_id":"fb7cb8d5-bda2-435c-8309-f202dde6e845","year":2018},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.477334Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:4f878d92a0ab6ee1ede5267f23600fb502f52a4accf99b78ab7ff76d6cf10294","observation_id":"bf7cea43-2d1b-484f-8642-ab3d228a0e81","resolution":{"observed_at":"2026-08-06T15:34:44.917281Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.904562Z","title":"Action recogni- tion with trajectory-pooled deep-convolutional descriptors","venue":null,"work_id":"97e67244-c2fe-48a2-9466-78529e192d7e","year":2015},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.480128Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:a75b554b9dba905f823c35c97cbb35dbd3b4f13cba53bf3c72524bb821b8f2c5","observation_id":"562445aa-e6ce-488a-ac70-6da7fe1054ed","resolution":{"observed_at":"2026-08-06T15:34:44.907511Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.894975Z","title":"Temporal segment net- works: Towards good practices for deep action recognition","venue":null,"work_id":"a374995b-5caf-4c34-ac11-678cac9b08d3","year":2016},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.482802Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:76b3c5fe030ae5331711cf3ed4af13f90223b85ab3cedcb4aecdf9207fd2d9c6","observation_id":"1ac8883f-6190-43e9-9dd7-56414c619435","resolution":{"observed_at":"2026-08-06T15:34:44.898103Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.884657Z","title":"Deep learning for video classification and captioning","venue":null,"work_id":"ff5496f1-3599-460c-8aa4-a3ecfc274d77","year":2017},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.485388Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:dd68f930acd35739fa71113eed54fc333b3a10b0a8adbe1597cc9fe8178c5d4b","observation_id":"2d08469b-7499-4da5-9a0c-291c211afaf6","resolution":{"observed_at":"2026-08-06T15:34:44.887994Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.875151Z","title":"Msr-vtt: A large video description dataset for bridging video and language","venue":null,"work_id":"2f6eacee-c797-4468-9d45-cb2342897154","year":2016},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.488052Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:7069e63ef31813842bee83a17882600e1ce67bc0bccc2e495d6a185e24b57675","observation_id":"04648ed4-742e-4fc7-a6dc-0c4f81a1dd1e","resolution":{"observed_at":"2026-08-06T15:34:44.878270Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16994","last_updated":"2024-04-29T14:52:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-04-25T19:29:55Z","title":"PLLaVA : Parameter-free LLaVA Extension from Images to Videos for Video Dense Captioning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16994","snapshot_observed_at":"2026-08-06T15:34:44.490784Z","title":"Pllava: Parameter-free llava extension from images to videos for video dense captioning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.490784Z"},"links":{"cited_paper":"/paper/2404.16994","citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:e311d2389d9b63e5199dc196cdf32e4f0b359106111c02f3ae74820cdb5daf53","observation_id":"62968b3d-c61b-44b4-b4b3-72671e84ee31","resolution":{"observed_at":"2026-08-06T15:34:44.490784Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.865283Z","title":"Beyond short snippets: Deep networks for video classification","venue":null,"work_id":"432275c7-c149-43c8-a619-23cf49d1797b","year":null},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.493641Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:5ed5dbde58ff6dda3b0c3c5f6f0046039a3515cfad0e013853efca04165a1e4b","observation_id":"39de8a1a-26a0-4798-a23c-7dd48455c721","resolution":{"observed_at":"2026-08-06T15:34:44.868532Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.07225","last_updated":"2022-10-13T17:50:24Z","snapshot_observed_at":"2026-08-09T12:39:06.378102Z","submitted_at":"2022-10-13T17:50:24Z","title":"Unified Vision and Language Prompt Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.07225","snapshot_observed_at":"2026-08-06T15:34:44.496621Z","title":"Unified vision and language prompt learning","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.496621Z"},"links":{"cited_paper":"/paper/2210.07225","citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:66c0bba81bf211a109ff05409696723f434f0bec71f1f890841b288d58bf2afe","observation_id":"624eca69-ece9-48ad-a3fe-6ee505763a9a","resolution":{"observed_at":"2026-08-06T15:34:44.496621Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.855290Z","title":"Sigmoid loss for language image pre-training","venue":null,"work_id":"1145331e-ef4e-4cf5-87d8-68181dcdfe01","year":2023},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.499401Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:c6edf97049297ebcb3d2bcae62e1cd80dae4163628df9ef6be55e86fbed27cb1","observation_id":"4e413645-f57d-425e-b8af-00f7ad24a065","resolution":{"observed_at":"2026-08-06T15:34:44.858675Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.843464Z","title":"Real-time action recognition with enhanced motion vector cnns","venue":null,"work_id":"2dba500b-96dc-46fc-81ce-a3b5c8f73d3f","year":null},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.502074Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:e86e82cb48d8515c2f679d2303a88170c0d13a01089889569ca29c01a03dc4e7","observation_id":"cadc5b69-4b8b-4b22-94c9-89979ec767a7","resolution":{"observed_at":"2026-08-06T15:34:44.848526Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.02858","last_updated":"2023-10-25T06:23:31Z","snapshot_observed_at":"2026-07-06T15:38:39.712379Z","submitted_at":"2023-06-05T13:17:27Z","title":"Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.02858","snapshot_observed_at":"2026-08-06T15:34:44.504615Z","title":"Video-llama: An instruction-tuned audio-visual language model for video un- derstanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.504615Z"},"links":{"cited_paper":"/paper/2306.02858","citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:c1191901f4f13d6b87a25846b5066ca9decd7c27026143995d0109477129200e","observation_id":"5bb3d3fb-30bd-42fc-8a91-94d8c352daf4","resolution":{"observed_at":"2026-08-06T15:34:44.504615Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.507428Z","title":"Conditional prompt learning for vision-language mod- els","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.507428Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:b01301b4e2f090b1045f4cc122f21e0f3eb25a38c183a43bbabd6111f4b5f520","observation_id":"2e549f92-9770-44e1-8e79-718f53c45890","resolution":{"observed_at":"2026-08-06T15:34:44.507428Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:34:44.510044Z","title":"Learning to prompt for vision-language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.510044Z"},"links":{"citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:715ace8823c5bbe4574f079af87e54f4b51c4504cb4fb93e78f1101bfde4f83b","observation_id":"3e308a2b-6870-4be2-8fd7-30105dfceff0","resolution":{"observed_at":"2026-08-06T15:34:44.510044Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.10592","last_updated":"2023-10-02T16:38:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-20T18:25:35Z","title":"MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.10592","snapshot_observed_at":"2026-08-06T15:34:44.512500Z","title":"Minigpt-4: Enhancing vision- language understanding with advanced large language mod- els","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-06T15:34:44.512500Z"},"links":{"cited_paper":"/paper/2304.10592","citing_paper":"/paper/2507.15569"},"observation_digest":"sha256:d45fc2fb7052589792af8e4d308704bdc4c3489917c6d4248cc5ca3e3d7c42c7","observation_id":"35094d80-6c56-4e63-8554-ee5219e9d0da","resolution":{"observed_at":"2026-08-06T15:34:44.512500Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2507.15569","last_updated":"2025-07-21T12:50:49Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-09T12:39:27.106704Z","submitted_at":"2025-07-21T12:50:49Z","title":"DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding"},"reference_resolution":{"displayed":57,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":25,"verified_exact":0,"verified_fuzzy":32},"total_outbound_references":57},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 57 of 57 outbound references and 0 inbound Pith citation observations for arXiv:2507.15569."}