{"as_of":"2026-08-23T14:44:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:6d3a7a1468bc57bf8f4b8abd4bd6a3da1148df9c6b4136ca2bac0738cb1cb01f","coverage":[{"denominator":51,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":51,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-01T18:11:04.192167Z","state":"measured"},{"denominator":51,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":51,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-23T06:30:58.430688+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2607.17386/citation-record","integrity":"/paper/2607.17386/integrity","json":"/paper/2607.17386/citation-record.json","paper":"/paper/2607.17386"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:00.691639Z","title":"One token to seg them all: Language instructed reasoning seg- mentation in videos.Advances in Neural Information Processing Systems, 37:6833–6859, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:00.691639Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:f63f686f77baf28152f8799081646e4060839de13eb3aa7e5e8800065f03277c","observation_id":"7c911503-5255-4be9-9c32-4a06b43b0c4a","resolution":{"observed_at":"2026-08-01T18:11:00.691639Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:00.744246Z","title":"Meteor: An automatic metric for mt evaluation with improved correlation with human judgments","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:00.744246Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:3dc9203f3a03ec0ee3765fde3103c3c02bce666d781bb689b21412c2160078c5","observation_id":"e726ec18-5a18-4e01-a66d-acf19c22c573","resolution":{"observed_at":"2026-08-01T18:11:00.744246Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:00.790110Z","title":"End-to-end referring video object segmentation with multimodal transformers","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:00.790110Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:7203983d7ed308a5ea1c5efb586a5486feafad9c05d5209543f7849df9230dc4","observation_id":"bc7c76ca-1343-4e91-9975-31bbadbdd027","resolution":{"observed_at":"2026-08-01T18:11:00.790110Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:00.854409Z","title":"Streamingtom: Streaming token compres- sion for efficient video understanding.arXiv preprint arXiv:2510.18269, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:00.854409Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:ee7d68087a76380056a2938969c8122ddd59bc2877a840426197bf5e0a22b6bb","observation_id":"ebeeb569-da85-4acb-8d79-5b96c294852e","resolution":{"observed_at":"2026-08-01T18:11:00.854409Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:00.919449Z","title":"Mevis: A large-scale benchmark for video segmentation with motion expressions","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:00.919449Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:336f67af46a6ba95e833e965f7d98a146efdfa9446ab0e317f8ffe99cbb0cbe2","observation_id":"aac304ac-f50d-4d06-89f3-224d7201ef3f","resolution":{"observed_at":"2026-08-01T18:11:00.919449Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:00.984738Z","title":"The unmanned aerial vehicle benchmark: Object detection and tracking","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:00.984738Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:31f1be5b05e019ce9df028e7199f3546020489b7da6fa20cbb880585c98593ec","observation_id":"f0495812-3451-4c19-85ed-ad679685c4c3","resolution":{"observed_at":"2026-08-01T18:11:00.984738Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:01.082341Z","title":"Framefusion: Combining similarity and importance for video token reduction on large vision language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:01.082341Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:de8c4ac32b75dc5d9dcecea5ce28e2b9eb549874c94cbff9212b9e081123acb3","observation_id":"f6ace37c-9484-43c0-8418-585971a9f62c","resolution":{"observed_at":"2026-08-01T18:11:01.082341Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.00318","last_updated":"2026-04-04T04:38:43Z","snapshot_observed_at":"2026-08-14T22:00:50.132781Z","submitted_at":"2025-05-31T00:08:21Z","title":"Chain-of-Frames: Advancing Video Understanding in Multimodal LLMs via Frame-Aware Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.00318","snapshot_observed_at":"2026-08-01T18:11:01.204751Z","title":"Chain-of-frames: Advancing video understanding in multimodal llms via frame-aware reasoning.arXiv preprint arXiv:2506.00318, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:01.204751Z"},"links":{"cited_paper":"/paper/2506.00318","citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:c7d825e6ef51730ef3d72c7783aeafd23f001af29cf80510ed93a17b890e1f3f","observation_id":"e7486384-4444-47a1-b64c-6ab04edcd40b","resolution":{"observed_at":"2026-08-01T18:11:01.204751Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:01.296336Z","title":"The devil is in temporal token: High quality video reasoning segmentation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:01.296336Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:ddf750b05eb13088118ae6732962d3daaff299b42922cb2a56ac3abf31ac25d8","observation_id":"954877f9-8f77-44ff-bdd2-48c34a27fa1d","resolution":{"observed_at":"2026-08-01T18:11:01.296336Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:01.367418Z","title":"Rsgpt: A remote sensing vision language model and benchmark.ISPRS Journal of Photogrammetry and Remote Sensing, 224:272–286, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:01.367418Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:bc7f0e6fcb027ce18ae21ad478d9bf84fa06c635b79dfa45fab1b8a5870f72ff","observation_id":"c2dae05a-a68c-4d7b-9543-b872f27f70de","resolution":{"observed_at":"2026-08-01T18:11:01.367418Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:01.457054Z","title":"Prunevid: Visual token pruning for efficient video large language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:01.457054Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:0a562f2de6e3dad58a729a326e6c0e6501b8127beb80d39beb78cfe13b47f344","observation_id":"69ff8e28-3474-4e55-8196-1d2bf37d04b2","resolution":{"observed_at":"2026-08-01T18:11:01.457054Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:01.521166Z","title":"Multi-granular spatio-temporal token merging for training-free acceleration of video llms","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:01.521166Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:f2d87db552c1fe15c747c0dbce9280562c558454bc4e5a5ebe07f14d93411dce","observation_id":"dfecb740-b81a-49cf-a800-cfeac3bca07f","resolution":{"observed_at":"2026-08-01T18:11:01.521166Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06234","last_updated":"2025-01-27T01:45:15Z","snapshot_observed_at":"2026-08-16T13:11:24.808006Z","submitted_at":"2024-10-08T17:45:51Z","title":"TEOChat: A Large Vision-Language Assistant for Temporal Earth Observation Data","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.06234","snapshot_observed_at":"2026-08-01T18:11:01.580630Z","title":"Teochat: A large vision-language assistant for temporal earth observation data.arXiv preprint arXiv:2410.06234, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:01.580630Z"},"links":{"cited_paper":"/paper/2410.06234","citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:5fb2af7d2c66c859edb6add5b078f5b4df9abfd571d6aa97c6928bc923c4fa30","observation_id":"c57608d3-d131-4995-8b1c-eff11246edac","resolution":{"observed_at":"2026-08-01T18:11:01.580630Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:01.696273Z","title":"Lisa: Reasoning segmentation via large language model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:01.696273Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:98b189639be2bf3514c2589594a00ea660b3a53974b7f9615cb93ee3a8923d2b","observation_id":"a485416f-d454-4b76-9ef5-7fc4ede0b8cf","resolution":{"observed_at":"2026-08-01T18:11:01.696273Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:01.761048Z","title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:01.761048Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:1f9fe34899b6d57203240524d2eae75137b5449b344b0419cf571fd650589997","observation_id":"83cd5b5c-dd0b-47d7-a5e5-ee9ef889736f","resolution":{"observed_at":"2026-08-01T18:11:01.761048Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:01.849069Z","title":"Referdino: Referring video object segmentation with visual grounding foundations","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:01.849069Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:7635ca276902410d0d4775e926160d44433fed34918c1c4a277c748c5000bf14","observation_id":"6731052e-dbfa-41ad-814f-7b040ed21374","resolution":{"observed_at":"2026-08-01T18:11:01.849069Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:01.971650Z","title":"Video-llava: Learning united visual representation by alignment before projection","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:01.971650Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:168f3efa753037b6630f2519655f77d1c132c00d27521b54a6d75dd9adf597b4","observation_id":"4f45cead-57fa-4193-934d-3aad75ef807a","resolution":{"observed_at":"2026-08-01T18:11:01.971650Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:02.034375Z","title":"Glus: Global-local reasoning unified into a single large language model for video segmentation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:02.034375Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:83495262884edd0777ac13cbcc602a9bcba4104f849f9adf4dbd5b03e5daa2d3","observation_id":"27c684a3-7358-48c6-886a-e007211e6294","resolution":{"observed_at":"2026-08-01T18:11:02.034375Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:02.108301Z","title":"Visual instruction tuning.Advances in neural information processing systems, 36:34892–34916, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:02.108301Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:47eceaae195b22ce159350357fc06840d88b3c984b06d55ba8e929a69f5c8850","observation_id":"1e1e3f91-2614-4465-aaeb-bde5dc3df7a4","resolution":{"observed_at":"2026-08-01T18:11:02.108301Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.05679","last_updated":"2024-12-10T02:23:30Z","snapshot_observed_at":"2026-08-15T18:52:04.714439Z","submitted_at":"2024-12-07T15:11:21Z","title":"RSUniVLM: A Unified Vision Language Model for Remote Sensing via Granularity-oriented Mixture of Experts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.05679","snapshot_observed_at":"2026-08-01T18:11:02.265647Z","title":"Rsunivlm: A unified vision language model for remote sensing via granularity-oriented mixture of experts.arXiv preprint arXiv:2412.05679, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:02.265647Z"},"links":{"cited_paper":"/paper/2412.05679","citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:ec8c799d101c97a2884f0d7fc87adb07ad8e165ae94b7997517178d0d70b01bd","observation_id":"bf66e690-c950-4d14-b84c-12030a074936","resolution":{"observed_at":"2026-08-01T18:11:02.265647Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.10100","last_updated":"2024-07-08T04:33:37Z","snapshot_observed_at":"2026-08-16T13:42:55.361342Z","submitted_at":"2024-06-14T14:57:07Z","title":"SkySenseGPT: A Fine-Grained Instruction Tuning Dataset and Model for Remote Sensing Vision-Language Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.10100","snapshot_observed_at":"2026-08-01T18:11:02.304478Z","title":"Skysensegpt: A fine-grained instruction tuning dataset and model for remote sensing vision-language understanding.arXiv preprint arXiv:2406.10100, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:02.304478Z"},"links":{"cited_paper":"/paper/2406.10100","citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:bab6a3e1c8e3ee15caf05cbf9edf41d0da25135d43c80085abf8c4d1c4fb7df4","observation_id":"7da24c0e-c78c-49c0-915a-672bacdb4bf6","resolution":{"observed_at":"2026-08-01T18:11:02.304478Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:02.369009Z","title":"Video-chatgpt: Towards detailed video understanding via large vision and language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:02.369009Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:48c25949f476f6b572cbde7b3e11ea5e2870059d6c94c20040df17456ec0901b","observation_id":"0ebe0656-7020-49d2-88e4-e00b58c04323","resolution":{"observed_at":"2026-08-01T18:11:02.369009Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:02.428042Z","title":"Videoglamm: A large multimodal model for pixel-level visual grounding in videos","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:02.428042Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:60ce7b99d7539f4f9deb3e8e93882b5d95cab0ec00be76d5dcdd7ac6c59a1153","observation_id":"b7b0d143-8b00-4c13-8bfd-e131ad13230a","resolution":{"observed_at":"2026-08-01T18:11:02.428042Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:02.501383Z","title":"Geopix: A multimodal large language model for pixel-level image understanding in remote sensing.IEEE Geoscience and Remote Sensing Magazine, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:02.501383Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:66302c906d6c046f746464885730d07f6db160cedfa33f68d0625fc4b0bd57f4","observation_id":"113d5682-d67d-40cb-b88c-a2e3a48dad5a","resolution":{"observed_at":"2026-08-01T18:11:02.501383Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:02.552720Z","title":"Vhm: Versatile and honest vision language model for remote sensing image analysis","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:02.552720Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:f4414a8f8a1362d110f8f1658e8037940e8c712ae2087684ac1d4685fbe33f15","observation_id":"65940e87-c7ca-4c12-b52c-4d4054028358","resolution":{"observed_at":"2026-08-01T18:11:02.552720Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:02.593053Z","title":"Llava++: extending visual capabilities with llama-3 and phi-3 (2024).URL https://github.com/mbzuai-oryx/LLaVA- pp, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:02.593053Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:c6c4c015d575db82da59357cb790c0042199e67650a316d75862131e295e9b65","observation_id":"9ddf7593-9019-48ce-b84c-fa5f2f6664a9","resolution":{"observed_at":"2026-08-01T18:11:02.593053Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:02.644608Z","title":"Glamm: Pixel grounding large multimodal model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:02.644608Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:d1872786e75ef61a1443181fe3f3fdf2849208a38a5ddf7ba457415e4c5a68a1","observation_id":"086ae9bd-1812-4e22-acf0-874e4e16621b","resolution":{"observed_at":"2026-08-01T18:11:02.644608Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00714","last_updated":"2024-10-28T16:37:57Z","snapshot_observed_at":"2026-07-06T18:55:41.459417Z","submitted_at":"2024-08-01T17:00:08Z","title":"SAM 2: Segment Anything in Images and Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00714","snapshot_observed_at":"2026-08-01T18:11:02.721970Z","title":"Sam 2: Segment anything in images and videos.arXiv preprint arXiv:2408.00714, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:02.721970Z"},"links":{"cited_paper":"/paper/2408.00714","citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:87eec5eebe6cc34e04c63811568cf9c50499fc9f488c440a3754e56e234e970a","observation_id":"fb476229-664f-4803-89e2-2fd277f44817","resolution":{"observed_at":"2026-08-01T18:11:02.721970Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:02.803839Z","title":"Pixellm: Pixel reasoning with large multimodal model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:02.803839Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:78788be39940daa3c0819463333b3fb7359b8dbe396d2168d4eb6732d56d33a4","observation_id":"eec54fbe-cb03-45c0-b031-d8fcb329e4d4","resolution":{"observed_at":"2026-08-01T18:11:02.803839Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:02.860093Z","title":"Moviechat: From dense token to sparse memory for long video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:02.860093Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:a8f85591f5b7d17c9946e7da76225861419aa2492fa36e95530e4ae2debf2a1a","observation_id":"867c4e28-917a-4a41-93be-31ea3459c95a","resolution":{"observed_at":"2026-08-01T18:11:02.860093Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:02.915593Z","title":"Earthdial: Turning multi-sensory earth observations to interactive dialogues","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:02.915593Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:2d02296e78501ae8dc3c751bbff578c8421ead4ed18fa926f5bfc9d2dc85f068","observation_id":"d83d7eb4-6c63-4448-94a0-952dadbdb274","resolution":{"observed_at":"2026-08-01T18:11:02.915593Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:02.973608Z","title":"Drone-based rgb-infrared cross-modality vehicle detection via uncertainty-aware learning.IEEE Transactions on Circuits and Systems for Video Technology, 32(10):6700–6713, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:02.973608Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:773e48a8a5fbd2889ec3ef2a068b6433da8f5c24b3c9d7cf84addf7d3d186f1a","observation_id":"52daab6d-b37e-48b3-9d2e-d8222f199e69","resolution":{"observed_at":"2026-08-01T18:11:02.973608Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:03.030943Z","title":"Adaptive keyframe sampling for long video understanding","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:03.030943Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:d90c62b28b1983caa2373b184fc58880bc7feb71d0be660600c2a072310eea3a","observation_id":"17863fb7-2119-4768-8b96-53666e87d913","resolution":{"observed_at":"2026-08-01T18:11:03.030943Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:03.074265Z","title":"Cider: Consensus-based image description evaluation","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:03.074265Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:f581828574b2504f5df14af676b6667f4d1f96398d3df66acfbf2369417401d7","observation_id":"4e899a43-c9ad-4aeb-9ac4-0626688eb142","resolution":{"observed_at":"2026-08-01T18:11:03.074265Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:03.124523Z","title":"Geollava-8k: scaling remote-sensing multimodal large language models to 8k resolution.arXiv preprint arXiv:2505.21375, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:03.124523Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:303d6c8f4d7edb7cb12519151f9e16c19ad0d88cf812decd811946164365705c","observation_id":"a24a7245-aef3-4bb0-aa6b-10394c3e1801","resolution":{"observed_at":"2026-08-01T18:11:03.124523Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:03.221579Z","title":"Instructseg: Unifying instructed visual segmentation with multi-modal large language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:03.221579Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:74d979b872cbb42eaa80ae01b53de6e71daeba233259c40912c405ddfb1f897c","observation_id":"2dd4ab56-621a-491c-bbf2-07dda40d7475","resolution":{"observed_at":"2026-08-01T18:11:03.221579Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:03.350493Z","title":"Longvlm: Efficient long video understanding via large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:03.350493Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:a24ad8c0c9c33c1b36a414fd62bc21d62bc09c44927910846e16ef53f8088978","observation_id":"aa94ce04-88b5-493b-9d0e-b29f1600024d","resolution":{"observed_at":"2026-08-01T18:11:03.350493Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:03.422889Z","title":"Language as queries for referring video object segmentation","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:03.422889Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:307703f6c5d0cc992cd3b57fa8990e498e8b89966356dfb0b093d338327f4051","observation_id":"1a711df3-dda3-4d74-9bfd-c7b3a4daab95","resolution":{"observed_at":"2026-08-01T18:11:03.422889Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15841","last_updated":"2024-09-15T05:00:18Z","snapshot_observed_at":"2026-08-16T13:32:02.889396Z","submitted_at":"2024-07-22T17:58:04Z","title":"SlowFast-LLaVA: A Strong Training-Free Baseline for Video Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15841","snapshot_observed_at":"2026-08-01T18:11:03.486399Z","title":"Slowfast-llava: A strong training-free baseline for video large language models.arXiv preprint arXiv:2407.15841, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:03.486399Z"},"links":{"cited_paper":"/paper/2407.15841","citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:04cfb1f5ce3ed985eb9e261e09f32f1894f9df8988c14fb5d810a23fa9a7ca11","observation_id":"9ce06e3c-e938-4b4b-ab75-acf4c575595c","resolution":{"observed_at":"2026-08-01T18:11:03.486399Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:03.576528Z","title":"Visa: Reasoning video object segmentation via large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:03.576528Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:8df8c0c99eeb4d0ec46875c40d27d79e3fd86360ddbdc3a0054b296f03f82d22","observation_id":"32849a1b-d5f7-4a20-b197-fc29806e3be2","resolution":{"observed_at":"2026-08-01T18:11:03.576528Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:03.579616Z","title":"Referred by multi-modality: A unified temporal transformer for video object segmentation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:03.579616Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:a6f1b90026c69d954c8b9639d804c2b46ad04dced8383bbd955388ad2f486e6a","observation_id":"126070c8-ffb6-4120-bcfd-8e74f9716739","resolution":{"observed_at":"2026-08-01T18:11:03.579616Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:03.612401Z","title":"Self-chained image-language model for video localization and question answering.Advances in Neural Information Processing Systems, 36:76749–76771, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:03.612401Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:932533cec99faee38ca7b4a926901d7d58f4f699265b8ae23a0e44ff98bf63b9","observation_id":"f4431f6f-4b71-442c-8f56-39b9b7774506","resolution":{"observed_at":"2026-08-01T18:11:03.612401Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03226","last_updated":"2025-03-28T03:19:52Z","snapshot_observed_at":"2026-08-20T09:36:35.686768Z","submitted_at":"2024-10-04T08:26:06Z","title":"Frame-Voyager: Learning to Query Frames for Video Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.03226","snapshot_observed_at":"2026-08-01T18:11:03.661242Z","title":"Frame-voyager: Learning to query frames for video large language models.arXiv preprint arXiv:2410.03226, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:03.661242Z"},"links":{"cited_paper":"/paper/2410.03226","citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:e2990621e4871b2ac0549ac6e892baa394aeb3179213cde62fcf3a4ae6ace8af","observation_id":"4a2ee8db-e65b-4904-8f56-7f723e0d407d","resolution":{"observed_at":"2026-08-01T18:11:03.661242Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.04001","last_updated":"2025-11-03T17:35:29Z","snapshot_observed_at":"2026-08-16T10:23:40.545862Z","submitted_at":"2025-01-07T18:58:54Z","title":"Sa2VA: Marrying SAM2 with LLaVA for Dense Grounded Understanding of Images and Videos","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.04001","snapshot_observed_at":"2026-08-01T18:11:03.743258Z","title":"Sa2va: Marrying sam2 with llava for dense grounded understanding of images and videos.arXiv preprint arXiv:2501.04001, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:03.743258Z"},"links":{"cited_paper":"/paper/2501.04001","citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:1afcd71c3be9c36aa846de92fdf4e7e7af40a7db6e44e14a5d690045ea3ecfc6","observation_id":"4817844c-a271-4236-a712-ef3db908dedc","resolution":{"observed_at":"2026-08-01T18:11:03.743258Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:03.814064Z","title":"Skyeyegpt: Unifying remote sensing vision- language tasks via instruction tuning with large language model.ISPRS Journal of Photogram- metry and Remote Sensing, 221:64–77, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:03.814064Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:70886dc2c33b42990bb53bde2a2cb38633dfb808d413a57f5f74856f3c6f871e","observation_id":"bb233953-67a4-4e45-9c3c-ed97252e8ef0","resolution":{"observed_at":"2026-08-01T18:11:03.814064Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:03.885115Z","title":"Video-llama: An instruction-tuned audio-visual language model for video understanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:03.885115Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:5746bb2b906a71d00d0aed7a6a828b0de191ea52a73ea18822159dfc6d238455","observation_id":"bd172da8-7b7c-4156-9dcf-d6053158fda9","resolution":{"observed_at":"2026-08-01T18:11:03.885115Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.12490","last_updated":"2025-03-16T12:48:17Z","snapshot_observed_at":"2026-08-16T21:32:57.071157Z","submitted_at":"2025-03-16T12:48:17Z","title":"GeoRSMLLM: A Multimodal Large Language Model for Vision-Language Tasks in Geoscience and Remote Sensing","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.12490","snapshot_observed_at":"2026-08-01T18:11:03.944636Z","title":"Georsmllm: A multimodal large language model for vision-language tasks in geoscience and remote sensing.arXiv preprint arXiv:2503.12490, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:03.944636Z"},"links":{"cited_paper":"/paper/2503.12490","citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:cef131d07663eee55e19d834bfecb68ba0127a8eefb2225749555e7359e9303a","observation_id":"0cd98b9a-d26e-4ea7-90ad-819a85d924ee","resolution":{"observed_at":"2026-08-01T18:11:03.944636Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:04.007847Z","title":"Tifre: Text-guided video frame reduction for efficient video multi-modal large language models.arXiv preprint arXiv:2602.08861, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:04.007847Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:0512b401b6e444cb0e2031f35de2f775c18288c2b346a22eab909005a5466ddf","observation_id":"a1b883f2-6312-4707-a5a4-da08e0549566","resolution":{"observed_at":"2026-08-01T18:11:04.007847Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:04.079355Z","title":"Reason: Reinforced causal search with information bottleneck for video understanding","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:04.079355Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:a9cc5dd95ed084794cc723d4dbda3ad967ba27ef7af5042af34d40465fd542bc","observation_id":"9860a9dd-5836-408c-b37e-fe9a0a0d5ded","resolution":{"observed_at":"2026-08-01T18:11:04.079355Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:04.130946Z","title":"Detection and tracking meet drones challenge.IEEE transactions on pattern analysis and machine intelligence, 44(11):7380–7399, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:04.130946Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:95bba688a192a4a03dff6e7c673e469cfd0389e4d0a0fdd30ee4116e8dc58996","observation_id":"890e5b91-6ce5-49a7-9e45-11915eb46598","resolution":{"observed_at":"2026-08-01T18:11:04.130946Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T18:11:04.192167Z","title":"Focus: Efficient keyframe selection for long video understanding.arXiv preprint arXiv:2510.27280, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-01T18:11:04.192167Z"},"links":{"citing_paper":"/paper/2607.17386"},"observation_digest":"sha256:be381d589c94a3fd81bf8f1946e4fdf5877fa62cfc807ed937dc6149f6c3d542","observation_id":"42c5d8b8-44ec-4295-ac63-564a1914705d","resolution":{"observed_at":"2026-08-01T18:11:04.192167Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2607.17386","last_updated":"2026-07-19T19:07:19Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-20T00:31:16.068115Z","submitted_at":"2026-07-19T19:07:19Z","title":"SkyVLaM: Multimodal Large Language Model for UAV Video Understanding in Remote Sensing"},"reference_resolution":{"displayed":51,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":51,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":51},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"thesis":"As of 23 August 2026, this Paper Citation Record lists 51 of 51 outbound references and 0 inbound Pith citation observations for arXiv:2607.17386."}