{"as_of":"2026-08-19T09:32:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:72504ccb15110646360e5af80799da34e6bb4c8a0eb2483213e7de1a3966005a","coverage":[{"denominator":42,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":42,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-29T22:31:35.967550Z","state":"measured"},{"denominator":42,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":42,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-19T06:32:44.657259+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2605.25784/citation-record","integrity":"/paper/2605.25784/integrity","json":"/paper/2605.25784/citation-record.json","paper":"/paper/2605.25784"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":"Danish, Muzammal Naseer, Abhijit Das, Salman Khan, and Fahad S","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:ba0d03a09bf923ace2f3863227bd4eef4a688ee7b7886e10a7904f5ac4783f3a","observation_id":"a9710184-b82d-4262-ab7b-456adb33bd58","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":"Rs-llava: A large vision-language model for joint captioning and question answering in remote sensing imagery.Remote Sensing, 16(9), 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:335c24ddc29a871751fd9826e1a2c84a015419de4e764801d0c8496649f072d4","observation_id":"db80ee94-e21b-4d87-8e9f-fb972420e6b7","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":"Vhm: Versatile and honest vision language model for remote sensing image analysis.Proceedings of the AAAI Conference on Artificial Intelligence, 39(6):6381–6388, Apr","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:a68b9a86a3bd30b33f26018e9fe1324f5a23c796cba0077946329aff8342a74c","observation_id":"69a6fc78-c8fc-4ad6-8bb5-c1f7e332e1b0","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":"Earthdial: Turning multi-sensory earth observations to interactive dialogues","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:ea5bb075b363cf353497095f722c21e762f6e61b5d0796676bc692fc3b0bb9f6","observation_id":"fd88225f-0411-464c-81ca-ef355ba69028","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":"Geopixel: Pixel grounding large multimodal model in remote sensing","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:5d07fb6c9a6c7850a73b63705ef80b900dfee4d8232ad0c60cb5167c1cdfd4c5","observation_id":"eb13fd66-ec77-4e0e-a773-5ca8a8147867","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":"Towards faithful reasoning in remote sensing: A perceptually- grounded geospatial chain-of-thought for vision-language models","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:ec3090b34a9046878b1d51860edc009b0aaaacfa53ac1230aa3983f07818d913","observation_id":"9e45e72d-487d-4df1-86e7-3471ebacfc41","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":"Re- motereasoner: Towards unifying geospatial reasoning workflow.Proceedings of the AAAI Conference on Artificial Intelligence, 40(14):11883–11891, Mar","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:a2cc0a7048a2d314611464e2c88c8fe87a6823cfec90442f7648ac57c2f21af0","observation_id":"c68b019a-cd52-4972-bc9e-2490ce6a4234","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":"Rsgpt: A remote sensing vision language model and benchmark.ISPRS Journal of Photogrammetry and Remote Sensing, 224:272–286, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:3b7d16dce1aca61d191f40a4335010718c44b01465690b71475aa57b27240042","observation_id":"920aacfd-8dd2-4054-998d-e81e7d93774d","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":"Hrvqa: A visual question answering benchmark for high-resolution aerial images.ISPRS Journal of Photogrammetry and Remote Sensing, 214:65–81, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:d77042cdc7972501a0718c10eb4939a37e776907e6fb311ef17be61e39aadabd","observation_id":"6165743a-5899-4b24-a303-642b93f383ad","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":"Earthvqa: Towards queryable earth via relational reasoning-based remote sensing visual question answering","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:4e3a7e1bded3f2f5ac62cc0cea38b134ccadc8339d1e78b4b5fe140a684fd79e","observation_id":"33aa3ca8-0cac-45d9-a8f0-e0c702d2337e","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":"Rsvlm-qa: A benchmark dataset for remote sensing vision language model-based question answering","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:bead238bffbcb0709b6dbc818f7bed4f93c2dc3b372cb061f20df5a3f55f88b1","observation_id":"54ce4f99-4263-4a5d-bca1-310c8fbfa380","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":"Vrsbench: A versatile vision-language benchmark dataset for remote sensing image understanding.Advances in Neural Information Processing Systems, 37:3229–3242, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:3b85a32c9e2b3202a6c7557b609e34b5b2505ba06f9c9f31ded78d48235b7eed","observation_id":"b620d4cc-ec0c-41cb-aeda-32cedbf8fb97","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":"Geobench-vlm: Benchmarking vision-language models for geospatial tasks","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:a603a287c9bfb9a57dfc3e6580119f61ab1a96b2094539cf826ffa374ccb03e9","observation_id":"3a3d84c6-8e8d-4dd2-a789-67008f50ee72","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":"CHOICE: Benchmarking the remote sensing capabilities of large vision-language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:db4a30f5c8b865d914d4e71317d5351f9baff123f825f3c4cdf04d17edef9d22","observation_id":"e03c66d8-3a8c-41d3-b802-4c1a847111e5","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:d04fc979431be645e554bbd204261eb03b15199436e56092856f32877f24e5e5","observation_id":"d6b6ac35-5da2-43e3-b795-520479d35500","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2602.07045","last_updated":"2026-05-14T08:15:47Z","snapshot_observed_at":"2026-08-12T13:46:50.835837Z","submitted_at":"2026-02-04T08:21:33Z","title":"VLRS-Bench: A Vision-Language Reasoning Benchmark for Remote Sensing","version":2},"cited_work":{"arxiv_id":"2602.07045","doi":null,"metadata_source":"pith","pith_arxiv_id":"2602.07045","snapshot_observed_at":"2026-06-29T22:34:01.403114Z","title":"VLRS-Bench: A Vision-Language Reasoning Benchmark for Remote Sensing","venue":"cs.CV","work_id":"63109d0a-582a-448f-b3bc-0a0255c36cd1","year":2026},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"cited_paper":"/paper/2602.07045","citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:4832d52528f8f7c0e159fb8c7d6da0a153ad9f7705cd15ca6a1f6363484dd1c6","observation_id":"554f7a26-98cf-4b78-bf78-90c9896d8468","resolution":{"observed_at":"2026-06-29T22:34:01.404724Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":"Geo3dvqa: Evaluating vision-language models for 3d geospatial reasoning from aerial imagery","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:0c1bf0c8747f4c61edafb9fe74f0e893d87ce33e4e91a49c1b3e8250a66af9b3","observation_id":"aa922583-856b-4b3b-a1c4-0ec955207f24","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":"A survey of image classification methods and techniques for improving classification performance.International journal of Remote sensing, 28(5):823–870, 2007","venue":null,"work_id":null,"year":2007},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:ff5933a7177a2ed1f0693d22961326a1f69dabe18235c1b355af44d2e681ba79","observation_id":"5742272d-782e-463c-a7f9-918d39bbf814","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":"Lidar data fusion to improve forest attribute estimates: A review.Current Forestry Reports, 10(4):281–297, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:fc9cf3849c432a3e8f830dddc507cb7a0a777ee949bb585e36b0e5b45c5b79f6","observation_id":"9f5ef635-d6af-48da-aa1e-124baa8bfade","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":"Fine classification of urban tree species based on uav-based rgb imagery and lidar data.Forests, 15(2):390, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:6e9ec855beb0c799ade94a524ca557d4f54d9f6015fbb458f005016fa15f1b66","observation_id":"0d5c8e59-a882-4ee7-8bcc-8350c791b956","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":"A deep-learning-based tree species classification for natural secondary forests using unmanned aerial vehicle hyperspectral images and lidar.Ecological Indicators, 159:111608, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:0810d4ddfc00a8e295979922a6ac923c0c7199fa564ad9183e9a970a2dacbe0b","observation_id":"82a2188c-661d-4f06-8992-96dbdcab3b89","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":"Mapping urban tree species by integrating canopy height model with multi-temporal sentinel-2 data.Remote Sensing, 17(5):790, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:29591649832e94815d6a4355665b569693a8108af85166e3391bbc1c8fc4b6b8","observation_id":"2a5ad11d-977f-4868-ac61-3fea4c923fc0","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":"Object-based tree species classification using airborne hyperspectral images and lidar data.Forests, 11(1):32, 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:bb86106181870eb34119741e625aca85bc31c8802201b1c219ae4efdd5b25a08","observation_id":"000c3e6c-9c99-4641-bcab-b8568d184721","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":"Remote sensing vision-language foundation models without annotations via ground remote alignment","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:c6463a7eb5609eaff0fc556467650ca7cae9d192d54f1374ccf601d0ee9c1f89","observation_id":"09e8fe42-6d9f-4223-b9fc-9b819749ea72","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":"Remoteclip: A vision language foundation model for remote sensing.IEEE Transactions on Geoscience and Remote Sensing, 62:1–16, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:28eebd834e3494dff0cfdf7f7f1a8eb0b5d01e63ccfab73053c8a41cb34c46d8","observation_id":"aca24033-253c-4c41-95f3-d9daf28d9b49","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":"Rs5m and georsclip: A large scale vision-language dataset and a large vision-language model for remote sensing.IEEE Transactions on Geoscience and Remote Sensing, pages 1–1, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:094380b966e59a2e0ba57bcd44a7c7a244cf6eaef94e3c701a55ca2dfd1caf5c","observation_id":"2a8b2060-de0b-4d59-9a16-fcb876093c5e","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":"Skyeyegpt: Unifying remote sensing vision-language tasks via instruction tuning with large language model.ISPRS Journal of Photogrammetry and Remote Sensing, 221:64–77, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:c6085485e989bedb0c6e3857c5a03c1d11d3a88836bd3829e247bcb3bee2eb46","observation_id":"37d85fe9-e54f-4dfa-a0d8-de429a5f26dc","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":"Lhrs-bot: Empowering remote sensing with vgi-enhanced large multimodal language model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:da757550a08ebae3cd456eb74557da8e97078f76efcfc0e14579be91556890a9","observation_id":"e2f55ba0-76a4-464a-8a3e-f0627a5ff283","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:3e8f60a052fc8a6e1315e860185a09e60400e18c06da308e25ed15782d4c3c67","observation_id":"ff019c0c-e82f-41d7-9f61-cb7b3883d9b3","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.21375","doi":"10.48550/arxiv.2505.21375","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Geollava-8k: Scaling remote-sensing multimodal large language models to 8k resolution","venue":"ArXiv.org","work_id":"e8b770f4-ab0f-4fa1-9b6e-f9bd0e4e8960","year":2025},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:50750a0bc5f36d8ae05fc549c10f0913bd24700509dd3a0c26cabfa1bca9b11e","observation_id":"07ebdcd6-573d-47df-9841-22d0ad1034e4","resolution":{"observed_at":"2026-06-29T22:34:01.402865Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":"Terramind: Large-scale generative multimodality for earth observation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:2c984b228de4ae320d7203da774fd46297c73ad4888ada74c63c11d4e891b98a","observation_id":"fec4b20c-25b4-4efd-8166-850809763eb6","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.14201","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:34:01.393453Z","title":"Geoeyes: On-demand visual focusing for evidence-grounded understanding of ultra-high-resolution re- mote sensing imagery","venue":null,"work_id":"0d316c52-7eb2-4906-8b3c-d6ff2c4d5135","year":2026},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:eb01e961cbcb40085e1016e4e608e1c4a07f267f7ad7e5732844738c5e425ff1","observation_id":"df02b4de-1ce3-480f-a6f2-8f28a4a83217","resolution":{"observed_at":"2026-06-29T22:34:01.396173Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.23828","last_updated":"2024-11-13T09:06:18Z","snapshot_observed_at":"2026-08-19T01:23:59.023305Z","submitted_at":"2024-10-31T11:20:13Z","title":"Show Me What and Where has Changed? Question Answering and Grounding for Remote Sensing Change Detection","version":2},"cited_work":{"arxiv_id":"2410.23828","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.23828","snapshot_observed_at":"2026-07-01T10:35:41.438606Z","title":"Show me what and where has changed? question answering and grounding for remote sensing change detection.arXiv preprint arXiv:2410.23828","venue":null,"work_id":"1833314a-a084-4f24-ad30-292f27b30b0b","year":2024},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"cited_paper":"/paper/2410.23828","citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:a8ca19e3424a010afde629705a4cd310aa21dce35a3f07f038a31feaaeeed17d","observation_id":"431b1348-a209-4380-8ca0-8b9e81987f37","resolution":{"observed_at":"2026-06-29T22:34:01.390806Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.12207","last_updated":"2025-08-13T05:17:53Z","snapshot_observed_at":"2026-08-19T03:12:21.170619Z","submitted_at":"2025-05-18T02:45:19Z","title":"Can Large Multimodal Models Understand Agricultural Scenes? Benchmarking with AgroMind","version":3},"cited_work":{"arxiv_id":"2505.12207","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.12207","snapshot_observed_at":"2026-06-29T22:34:01.406814Z","title":"Can large multimodal models understand agricultural scenes? benchmarking with agromind","venue":null,"work_id":"1b80af65-9017-49a0-afb7-eda946d37088","year":2025},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"cited_paper":"/paper/2505.12207","citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:0456b15b4d0f84d4bfc137bef3d6ad45b87fbf8fbc1c0b1d0d29bd90c71f1963","observation_id":"55f06c66-96e7-41a9-8ace-4dffebf7d04b","resolution":{"observed_at":"2026-06-29T22:34:01.409852Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":"Canopy height model and naip imagery pairs across conus.Scientific Data, 12(1):322, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:6e8998c8847caa3a627d3ef03335a99f01db67d73f8a87000afe879f82a2437d","observation_id":"b3346d17-55b3-4617-9efe-2d1a599c5179","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:24e42bd01975e44eba30463850901179681654886c1e989bb808c8da540d779c","observation_id":"053b82ac-958a-490d-a94d-87afcb857561","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":"Openai gpt-5 system card, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:cf8631ba2141c85b2d5d689a8094438f70ca441d9aed777f3aaee04dc7d87acc","observation_id":"e1d1daf3-6afe-487e-9026-45a489415991","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":"Qwen3.5: Accelerating productivity with native multimodal agents, February 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:2e77d3b5ebb0b89dbd0b0afab817a0bf88c5bbf0da0a3aa62c693124c28e9726","observation_id":"22fdcaac-1ac2-4cdf-8e53-707bba657b2e","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":"Internvl3.5: Advancing open-source multimodal models in versatility, reasoning, and efficiency, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:a4e560c93670965f0ffed8694810a2ab11adb4289e064ef6d10e213dbfc877e7","observation_id":"6f2bc2d7-0d8c-47c4-82a4-524a6f2fe106","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":"Mistralai/mistral-small-3.1-24b-instruct-2503 · hugging face","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:346976ebf930dd945bf3894c45733ba5aec9b53e4b0727c0345e41647b9acd35","observation_id":"b3107960-a421-4df7-a5f5-c27b230b988f","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":"Phi-4 technical report, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:068273feb72f4363786cbcb46ca4b8e27e3c3fbe1697075902962e87a8be3066","observation_id":"ff6d0480-5600-457e-a755-3861ddd6c250","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T22:31:35.967550Z","title":"Gemma 3 technical report, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-06-29T22:31:35.967550Z"},"links":{"citing_paper":"/paper/2605.25784"},"observation_digest":"sha256:dc7bb8cae80827b3438a60c14e8225ef77e57623c3978fa00b3ed6f76d3d2804","observation_id":"9e6f50ea-c3cd-4714-9e7f-3f5d16218d51","resolution":{"observed_at":"2026-06-29T22:31:35.967550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2605.25784","last_updated":"2026-05-25T12:30:33Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-13T15:50:32.401014Z","submitted_at":"2026-05-25T12:30:33Z","title":"VertiCue-Bench: Diagnosing Whether MLLMs Use Height Cues to Resolve 2D Ambiguity in Remote Sensing Natural Scenes"},"reference_resolution":{"displayed":42,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":37,"verified_exact":5,"verified_fuzzy":0},"total_outbound_references":42},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"thesis":"As of 19 August 2026, this Paper Citation Record lists 42 of 42 outbound references and 0 inbound Pith citation observations for arXiv:2605.25784."}