{"as_of":"2026-08-15T11:06:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:0e3a49d9e100c6943a7f816db9d061a5b26e58f32c3c2a8de2da3df3271affc2","coverage":[{"denominator":66,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":66,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T21:52:25.446679Z","state":"measured"},{"denominator":68,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":68,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-15T06:32:42.880941+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-28T06:58:39.117228Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-02T07:26:45.809288Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"cited_work":{"arxiv_id":"2506.23219","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.23219","snapshot_observed_at":"2026-07-02T07:26:45.809288Z","title":"https://api.semanticscholar.org/CorpusID: 280010693","venue":null,"work_id":"99e1ebd3-1e7b-4a41-a793-35ef7ac42029","year":2025},"citing_paper":{"arxiv_id":"2604.08033","last_updated":"2026-04-09T09:38:15Z","snapshot_observed_at":"2026-08-14T05:37:55.126987Z","submitted_at":"2026-04-09T09:38:15Z","title":"IoT-Brain: Grounding LLMs for Semantic-Spatial Sensor Scheduling","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-10T17:49:16.409834Z"},"links":{"cited_paper":"/paper/2506.23219","citing_paper":"/paper/2604.08033"},"observation_digest":"sha256:17a1eeb439b4826dbe6a9913de7a4223e304ac68e1c4958b9d5bca0d220d0435","observation_id":"eba8392f-a4ca-4c0c-b565-bdf2db73376e","resolution":{"observed_at":"2026-05-11T06:05:56.875660Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"cited_work":{"arxiv_id":"2506.23219","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.23219","snapshot_observed_at":"2026-07-02T07:26:45.809288Z","title":"https://api.semanticscholar.org/CorpusID: 280010693","venue":null,"work_id":"99e1ebd3-1e7b-4a41-a793-35ef7ac42029","year":2025},"citing_paper":{"arxiv_id":"2606.04381","last_updated":"2026-06-03T02:54:59Z","snapshot_observed_at":"2026-07-06T23:44:28.265478Z","submitted_at":"2026-06-03T02:54:59Z","title":"From Symbolic to Geometric: Enabling Spatial Reasoning in Large Language Models","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-28T06:58:39.117228Z"},"links":{"cited_paper":"/paper/2506.23219","citing_paper":"/paper/2606.04381"},"observation_digest":"sha256:58b80630ebc878114268268684ba316140bde36c61b021ec57e78dac66132c10","observation_id":"4a8bd704-34ac-4f67-ab3b-97c1b8c68f6c","resolution":{"observed_at":"2026-07-02T07:26:45.810959Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.23219/citation-record","integrity":"/paper/2506.23219/integrity","json":"/paper/2506.23219/citation-record.json","paper":"/paper/2506.23219"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2403.09059","last_updated":"2024-11-12T06:15:50Z","snapshot_observed_at":"2026-08-13T00:54:39.534027Z","submitted_at":"2024-03-14T02:56:38Z","title":"LAMP: A Language Model on the Map","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.09059","snapshot_observed_at":"2026-08-06T21:52:19.753815Z","title":"Lamp: A language model on the map","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:19.753815Z"},"links":{"cited_paper":"/paper/2403.09059","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:783122e415791c68ef6038c89fd91dd414e7cb966de00b87d4a2bbc7af2cd9fb","observation_id":"a66870f0-4af1-4c6e-b323-672ffb45e0fd","resolution":{"observed_at":"2026-08-06T21:52:19.753815Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:34.808717Z","title":"City foundation models for learning general purpose rep- resentations from openstreetmap","venue":null,"work_id":"d0cff2ae-7aba-457d-a277-7f67f1acbd42","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:19.839179Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:6d1dfbf8073b79187d49319eaf479ac4e6b3cf1e18417018498608274a9480a5","observation_id":"69493768-4882-459e-b673-245ff3654219","resolution":{"observed_at":"2026-08-06T21:52:34.883775Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:34.620458Z","title":"Street view imagery in urban analytics and gis: A review","venue":null,"work_id":"cac9e22e-a157-4c47-8bfb-f1d3da2eba56","year":2021},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:19.924010Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:d03bb456c2c5a9e809b0de5ef40fffe6b3892a38408c19d5151f3b581eda9968","observation_id":"983a5a70-1d8b-48e3-99a8-3f7c8161bcbc","resolution":{"observed_at":"2026-08-06T21:52:34.689303Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15818","last_updated":"2023-07-28T21:18:02Z","snapshot_observed_at":"2026-08-02T16:17:50.621617Z","submitted_at":"2023-07-28T21:18:02Z","title":"RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15818","snapshot_observed_at":"2026-08-06T21:52:20.060217Z","title":"Rt-2: Vision-language-action models transfer web knowledge to robotic control","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:20.060217Z"},"links":{"cited_paper":"/paper/2307.15818","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:d63055eb7fa862727d32269d5339882db7e43a807a2d4c5fe4898b695b4c5530","observation_id":"10e1defb-e8da-4ac3-88bc-9edc797d602c","resolution":{"observed_at":"2026-08-06T21:52:20.060217Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:34.427094Z","title":"Touchdown: Natural language naviga- tion and spatial reasoning in visual street environments","venue":null,"work_id":"93ed5db4-3b79-4cf6-b4f7-ae90cf4e2c00","year":2019},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:20.178492Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:f28f6f2653508f2c727ef7e32b9f495174b7ffc80300c1059062dc8fb45e10a2","observation_id":"bfce9a89-df1e-437e-b8b9-a9c1ed8543ae","resolution":{"observed_at":"2026-08-06T21:52:34.487883Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.12793","last_updated":"2023-11-28T08:52:50Z","snapshot_observed_at":"2026-08-14T06:42:43.375489Z","submitted_at":"2023-11-21T18:58:11Z","title":"ShareGPT4V: Improving Large Multi-Modal Models with Better Captions","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.12793","snapshot_observed_at":"2026-08-06T21:52:20.279386Z","title":"Sharegpt4v: Improving large multi-modal models with better captions","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:20.279386Z"},"links":{"cited_paper":"/paper/2311.12793","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:9a65c29f1cb36c64d98a30952c8c5cf58349940df3ae7fae657e1dc66b5d6596","observation_id":"c0282037-4302-4a00-acf3-f66c822daa7d","resolution":{"observed_at":"2026-08-06T21:52:20.279386Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16821","last_updated":"2024-04-29T20:24:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-04-25T17:59:19Z","title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16821","snapshot_observed_at":"2026-08-06T21:52:20.373110Z","title":"How far are we to gpt-4v? closing the gap to commercial multimodal models with open-source suites","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:20.373110Z"},"links":{"cited_paper":"/paper/2404.16821","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:76c2e75260aee7eb315a1dfc054475fc5b53ee44506a61fd5237949cac3e066b","observation_id":"f5b2046a-852f-4c85-b199-852e4af903d3","resolution":{"observed_at":"2026-08-06T21:52:20.373110Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:34.267818Z","title":"Internvl: Scaling up vision founda- tion models and aligning for generic visual-linguistic tasks","venue":null,"work_id":"576176a2-1fec-4d0e-878f-acb8121dfea7","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:20.462260Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:f9bd97306697d377907631f7c805eb5b583a887945c077da84673c57a547d869","observation_id":"44fc0a9b-b231-4285-8142-b90368f7e855","resolution":{"observed_at":"2026-08-06T21:52:34.338270Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01584","last_updated":"2024-10-15T01:16:20Z","snapshot_observed_at":"2026-08-14T09:35:24.079025Z","submitted_at":"2024-06-03T17:59:06Z","title":"SpatialRGPT: Grounded Spatial Reasoning in Vision Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01584","snapshot_observed_at":"2026-08-06T21:52:20.579394Z","title":"Spatial- rgpt: Grounded spatial reasoning in vision language model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:20.579394Z"},"links":{"cited_paper":"/paper/2406.01584","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:3460e52df0fa887bcbb58cfc944d69514919f31f8f8bbd2e90ecdcb2b8e1a4a3","observation_id":"ca1bd479-40fe-484f-a892-1beea35e5999","resolution":{"observed_at":"2026-08-06T21:52:20.579394Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:20.691552Z","title":"Understanding world or predict- ing future? a comprehensive survey of world models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:20.691552Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:3cbf055aa4e404cce5eb26f33e2e645809a124940e18cbad9dd356d53a1e4fc2","observation_id":"d5635dfc-a5bd-429a-a4ca-67bd4b671ad8","resolution":{"observed_at":"2026-08-06T21:52:20.691552Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.14233","last_updated":"2023-05-23T16:49:14Z","snapshot_observed_at":"2026-08-14T22:24:15.551461Z","submitted_at":"2023-05-23T16:49:14Z","title":"Enhancing Chat Language Models by Scaling High-quality Instructional Conversations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.14233","snapshot_observed_at":"2026-08-06T21:52:20.776159Z","title":"Enhancing chat language models by scal- ing high-quality instructional conversations","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:20.776159Z"},"links":{"cited_paper":"/paper/2305.14233","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:29784efe30e85036aa26719d13a018d2652d79b637d4aee8bc21e9164d2cdd9e","observation_id":"f8f69a58-ccc7-48f3-94c7-4d997701a538","resolution":{"observed_at":"2026-08-06T21:52:20.776159Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.05492","last_updated":"2024-06-07T15:51:08Z","snapshot_observed_at":"2026-08-13T05:54:09.385656Z","submitted_at":"2023-10-09T07:56:16Z","title":"How Abilities in Large Language Models are Affected by Supervised Fine-tuning Data Composition","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.05492","snapshot_observed_at":"2026-08-06T21:52:20.860831Z","title":"How abilities in large lan- guage models are affected by supervised fine-tuning data composition","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:20.860831Z"},"links":{"cited_paper":"/paper/2310.05492","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:0994fca7c8a89decc2d17dabe96b32e014f5817b81e8b87da38823b02e3deb2c","observation_id":"b09e2d68-1545-4e28-b366-9000934f1574","resolution":{"observed_at":"2026-08-06T21:52:20.860831Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:20.969557Z","title":"Vlmevalkit: An open- source toolkit for evaluating large multi-modality models,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:20.969557Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:a776823d5839b1f1b1351b03a200d81c9a50bd893091e50adb8a17b8bdb318e0","observation_id":"33b27999-3ad1-4f5a-ad15-fbb09880377a","resolution":{"observed_at":"2026-08-06T21:52:20.969557Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:34.093097Z","title":"Urban visual intelligence: Uncovering hidden city pro- files with street view images","venue":null,"work_id":"6ac9549e-e54e-4a39-92e5-2718f65f7ba9","year":2023},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:21.035326Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:b7b2ccbb624504535669d3ff3c7b66ceabb398d2ff1bda9b2ac75c3a3767241d","observation_id":"1b813eca-e9b5-4467-99db-6e324c9003b4","resolution":{"observed_at":"2026-08-06T21:52:34.158394Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:33.948118Z","title":"Agent- move: A large language model based agentic framework for zero-shot next location prediction","venue":null,"work_id":"fec6fe57-5d3f-43f6-b9b5-affb50ca1e37","year":2025},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:21.152105Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:0f996d94fe26a03b60eb457032a9de13b0ac78124c11a0f0be2e72320d8fa559","observation_id":"d719cf63-a281-4dc5-b9a6-fbf82669f95c","resolution":{"observed_at":"2026-08-06T21:52:34.016597Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:33.693058Z","title":"Citygpt: Empowering urban spatial cognition of large language models","venue":null,"work_id":"2706224d-90c5-458e-aebc-6c744045843a","year":2025},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:21.237329Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:109c58a9d82a26beae9705e260c5761c5f6a3eda34abf271db3f1cd922b528b6","observation_id":"5ab581ac-0389-4ed1-a2ef-94ca7df1bca1","resolution":{"observed_at":"2026-08-06T21:52:33.818015Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.09848","last_updated":"2025-04-14T03:38:31Z","snapshot_observed_at":"2026-08-09T21:17:35.287262Z","submitted_at":"2025-04-14T03:38:31Z","title":"A Survey of Large Language Model-Powered Spatial Intelligence Across Scales: Advances in Embodied Agents, Smart Cities, and Earth Science","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.09848","snapshot_observed_at":"2026-08-06T21:52:21.335105Z","title":"A survey of large language model-powered spatial intelligence across scales: Advances in embodied agents, smart cities, and earth science","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:21.335105Z"},"links":{"cited_paper":"/paper/2504.09848","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:0f4aa661283c83deb66a9eba784d0b1ffcc0ddbd84bd591213d6ba16024b216c","observation_id":"a030dcfb-3b62-4f79-8cd5-b941de0a9c60","resolution":{"observed_at":"2026-08-06T21:52:21.335105Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:33.500695Z","title":"City- bench: Evaluating the capabilities of large language models for urban tasks","venue":null,"work_id":"32e0dbbb-c9e0-4bb0-84a5-f1f6243d570f","year":2025},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:21.403131Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:5f847d868cf18c36766b06771c94526a50cb237bb6f135817dcd56a685d9bc35","observation_id":"50e01ba8-4998-46e5-b4d7-22fb6e6410dc","resolution":{"observed_at":"2026-08-06T21:52:33.556275Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:21.502930Z","title":"Imagebind: One embedding space to bind them all","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:21.502930Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:5ca88994028d405797cd3e5cf05be4015d54d7805aa9346fbd58536e7ed6d2a2","observation_id":"dc817789-0086-487f-a05f-52dacc81f7ab","resolution":{"observed_at":"2026-08-06T21:52:21.502930Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.00823","last_updated":"2024-10-29T01:58:06Z","snapshot_observed_at":"2026-08-13T15:27:49.527306Z","submitted_at":"2024-10-29T01:58:06Z","title":"Mobility-LLM: Learning Visiting Intentions and Travel Preferences from Human Mobility Data with Large Language Models","version":1},"cited_work":{"arxiv_id":"2411.00823","doi":null,"metadata_source":"pith","pith_arxiv_id":"2411.00823","snapshot_observed_at":"2026-08-06T21:52:25.934068Z","title":"Mobility-LLM: Learning Visiting Intentions and Travel Preferences from Human Mobility Data with Large Language Models","venue":"cs.LG","work_id":"40f599c5-4b46-4b81-84c1-dc7f3ca8e6ef","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:21.581742Z"},"links":{"cited_paper":"/paper/2411.00823","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:fc13593573543834763f4422fc178fd228f45098759572c01d71d32839b8fa17","observation_id":"4f1933ac-cb85-483b-b11f-7834f2a98765","resolution":{"observed_at":"2026-08-06T21:52:26.087467Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:33.295368Z","title":"Regiongpt: Towards region understanding vision lan- guage model","venue":null,"work_id":"0edebe3b-1c7f-4975-9d4e-e0b44fcd0eba","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:21.665075Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:eb54f49176243378a800a73548231e9bcf6628a130878eb2d2b4d0a832349e4a","observation_id":"f61311cd-f3e4-4df6-b205-1b23b4838077","resolution":{"observed_at":"2026-08-06T21:52:33.357391Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16831","last_updated":"2025-01-22T08:45:56Z","snapshot_observed_at":"2026-08-13T00:45:37.091670Z","submitted_at":"2024-03-25T14:57:18Z","title":"UrbanVLP: Multi-Granularity Vision-Language Pretraining for Urban Socioeconomic Indicator Prediction","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.16831","snapshot_observed_at":"2026-08-06T21:52:21.758851Z","title":"Urbanvlp: A multi- granularity vision-language pre-trained foundation model for urban indicator prediction","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:21.758851Z"},"links":{"cited_paper":"/paper/2403.16831","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:eb03f5678e9b92d9501ed97bd9385fdb75c5d8e6b3239facf0639e1f2876d7c1","observation_id":"57e99a67-4c0a-4127-8b1c-d71f82485cd5","resolution":{"observed_at":"2026-08-06T21:52:21.758851Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:33.084582Z","title":"Vision-language models for medical report generation and visual question answering: A review, 2024","venue":null,"work_id":"b70cb0f6-9855-4790-9124-24b6dc204258","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:21.863618Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:2c1d32fbb6cefad13a48c2691030a1896a065d5fc943b7344746d7bc43bfa3b3","observation_id":"8e59aa17-e176-46fe-b9ea-ac00f6d93b75","resolution":{"observed_at":"2026-08-06T21:52:33.222258Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15266","last_updated":"2023-07-28T02:23:35Z","snapshot_observed_at":"2026-08-13T10:45:48.484753Z","submitted_at":"2023-07-28T02:23:35Z","title":"RSGPT: A Remote Sensing Vision Language Model and Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15266","snapshot_observed_at":"2026-08-06T21:52:21.928081Z","title":"Rsgpt: A remote sensing vision language model and benchmark","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:21.928081Z"},"links":{"cited_paper":"/paper/2307.15266","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:0a89eec55617c3d9af3c8cf8d24fbd7af880f849027f7dea9c02488c8c0b9c80","observation_id":"89386da8-4e18-4d7c-801e-a4d5714a0262","resolution":{"observed_at":"2026-08-06T21:52:21.928081Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01728","last_updated":"2024-01-29T06:27:53Z","snapshot_observed_at":"2026-08-13T07:00:35.291225Z","submitted_at":"2023-10-03T01:31:25Z","title":"Time-LLM: Time Series Forecasting by Reprogramming Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01728","snapshot_observed_at":"2026-08-06T21:52:22.043641Z","title":"Time-llm: Time series forecasting by reprogramming large language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:22.043641Z"},"links":{"cited_paper":"/paper/2310.01728","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:e44e206b4a50d5f49a20460d571cced31f31d311ef7377c0936a449bb1ff159c","observation_id":"4c330697-6e82-442f-9a38-205e3a50ed8c","resolution":{"observed_at":"2026-08-06T21:52:22.043641Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:32.910871Z","title":"Geochat: Grounded large vision-language model for remote sensing","venue":null,"work_id":"885dbdc0-ecf6-458e-849d-ab12815ed80c","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:22.119635Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:b9cca552a88b9fba2ae8423f1f3698622f48a8674a1e44c52cb0fe46551f298b","observation_id":"969e79a5-5099-4815-b454-9f78d05a7ada","resolution":{"observed_at":"2026-08-06T21:52:32.998699Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:32.723289Z","title":"Llava-med: Training a large language- and-vision assistant for biomedicine in one day.Advances in Neural Information Processing Systems, 36, 2024","venue":null,"work_id":"234ff8e3-f1f5-45e0-9f3b-6c56d84498ed","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:22.243892Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:ae05af29f7d8d9f5f320156b6047d57d7de2abb1bafa2f5c1153648501f8437c","observation_id":"1899628e-54f6-4046-8779-ac2b441d92f0","resolution":{"observed_at":"2026-08-06T21:52:32.803628Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:32.539200Z","title":"Urbangpt: Spatio- temporal large language models","venue":null,"work_id":"660b244a-3cbe-40ca-878e-2ebe09e5f0e7","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:22.380917Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:5326d7c10f7a5b3825855ecb59e2cc9f9068c472d4b7be3c21e1eecc1c765a20","observation_id":"0a1756f7-099d-4bc0-81f1-af3980b310af","resolution":{"observed_at":"2026-08-06T21:52:32.628393Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:32.366052Z","title":"Vila: On pre-training for visual language models","venue":null,"work_id":"a27edd05-2a5d-40ba-bc9d-5bd99e9dfd28","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:22.486258Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:268c0a95a9eb88039d2824fa80d7fea8f584d5e5e176624962c1042baac06592","observation_id":"7fcd0e7e-db48-4ec4-8f6f-70a57a09de3e","resolution":{"observed_at":"2026-08-06T21:52:32.420034Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:32.185801Z","title":"Improved baselines with visual instruction tuning","venue":null,"work_id":"910fbf88-58ad-48dc-b5e8-35fddb27d548","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:22.603566Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:d14532b8ac35bd9386acb7336b82f7b81346400973c03b440f3a1ffb3bb815b0","observation_id":"e414ca83-90fd-4e0b-a0ae-fe4fa6fc5785","resolution":{"observed_at":"2026-08-06T21:52:32.256034Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:32.005820Z","title":"Visual instruction tuning","venue":null,"work_id":"6cba9273-53f4-4278-8dff-af868cce82f5","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:22.687759Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:46e0c1077a3399b9feb407be25ca479fa7648a8689854d1249cc7a47e932177b","observation_id":"774b3427-2c81-4631-9e66-edd2016f0633","resolution":{"observed_at":"2026-08-06T21:52:32.082401Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:22.798308Z","title":"Citylens: Bench- 10 marking large language-vision models for urban socioeco- nomic sensing","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:22.798308Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:6eef582b0e96dabf27d0e1dddf84abe91abf0bc2c2630d3f8aea95a76edf29e4","observation_id":"e39f3b66-cf9c-4426-963e-f6fafe0362c4","resolution":{"observed_at":"2026-08-06T21:52:22.798308Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.10100","last_updated":"2024-07-08T04:33:37Z","snapshot_observed_at":"2026-08-12T23:42:36.797128Z","submitted_at":"2024-06-14T14:57:07Z","title":"SkySenseGPT: A Fine-Grained Instruction Tuning Dataset and Model for Remote Sensing Vision-Language Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.10100","snapshot_observed_at":"2026-08-06T21:52:22.909176Z","title":"Skysensegpt: A fine-grained in- struction tuning dataset and model for remote sensing vision- language understanding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:22.909176Z"},"links":{"cited_paper":"/paper/2406.10100","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:141832ec622e876d96f6eab3e87de4c58c17feddb50330b9bf6dd1a146b26147","observation_id":"cf9c645e-b765-4ad7-85d1-1def9183b5d5","resolution":{"observed_at":"2026-08-06T21:52:22.909176Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.00438","last_updated":"2023-12-01T09:10:33Z","snapshot_observed_at":"2026-08-13T05:12:36.592290Z","submitted_at":"2023-12-01T09:10:33Z","title":"Dolphins: Multimodal Language Model for Driving","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.00438","snapshot_observed_at":"2026-08-06T21:52:23.026538Z","title":"Dolphins: Multimodal language model for driving","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:23.026538Z"},"links":{"cited_paper":"/paper/2312.00438","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:9647c5a11585de4d535fed1426069697e260bcf0cbcc1513d535e99c1a54c13e","observation_id":"6dd443bf-64e6-47c3-be1f-8389e6fb831e","resolution":{"observed_at":"2026-08-06T21:52:23.026538Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:31.834802Z","title":"On the opportunities and chal- lenges of foundation models for geoai (vision paper)","venue":null,"work_id":"91cf7ae0-6b4f-4676-a997-f966b6937573","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:23.139438Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:b3489009275f8fdb8b954d47146496f0e090cc0e1a761b4d7102fa38a15f7e2d","observation_id":"335ccf75-0269-4d48-831b-104bd6031418","resolution":{"observed_at":"2026-08-06T21:52:31.920846Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:31.631802Z","title":"LLaMA 3.2: Advancing Vision, Edge, and Mo- bile Devices","venue":null,"work_id":"0218bf09-5753-406c-a3f1-e80412093b59","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:23.201817Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:104e981c8d155cdfa397ae1c5f1e001b287d3f3621b00d304514412794234e66","observation_id":"188d55e0-3b17-4ea4-bc96-86149bdc97ce","resolution":{"observed_at":"2026-08-06T21:52:31.727923Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.02544","last_updated":"2024-07-16T01:40:34Z","snapshot_observed_at":"2026-08-13T04:27:18.680428Z","submitted_at":"2024-02-04T15:46:43Z","title":"LHRS-Bot: Empowering Remote Sensing with VGI-Enhanced Large Multimodal Language Model","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.02544","snapshot_observed_at":"2026-08-06T21:52:23.293073Z","title":"Lhrs-bot: Empowering remote sensing with vgi-enhanced large multimodal language model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:23.293073Z"},"links":{"cited_paper":"/paper/2402.02544","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:1ab32530c744e045c6fc2e5dd182505f600b37def15afbeb98df9a1fb224943a","observation_id":"2bc776dd-464a-45e3-b098-c2300c8c73d3","resolution":{"observed_at":"2026-08-06T21:52:23.293073Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:31.423724Z","title":"Introducing chatgpt","venue":null,"work_id":"5fb89827-bb14-40a0-b112-c1ce56b911c1","year":2022},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:23.383898Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:4808ee051173ba069508a34ffeabb560ff0f45b9e9fc39a15e84fd96bb3d9fc8","observation_id":"bccbedbe-3ea0-47a8-a30b-31564a8b6581","resolution":{"observed_at":"2026-08-06T21:52:31.520569Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:31.220882Z","title":"Gpt-4v(ision) system card","venue":null,"work_id":"e3f97a34-3536-4159-b6ed-781d4bc828cb","year":2023},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:23.464642Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:ecdfbf2e83cfb5e32642c9824fcf8f907b0827ef1425d428509ea23d1ee126d9","observation_id":"8013e578-b61f-4c03-8585-db60f194e6e8","resolution":{"observed_at":"2026-08-06T21:52:31.306992Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:31.035922Z","title":"Hello GPT-4","venue":null,"work_id":"0a607a22-3132-4584-8897-839e5c5f92e6","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:23.579833Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:1024e4f064aab980482ea18c01699155a69dc784a8cfa5df34babb35afacdc56","observation_id":"35a0cb28-947f-4e9c-842a-a0b494a6e2ea","resolution":{"observed_at":"2026-08-06T21:52:31.119217Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12191","last_updated":"2024-10-03T15:54:49Z","snapshot_observed_at":"2026-08-06T05:35:29.109022Z","submitted_at":"2024-09-18T17:59:32Z","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12191","snapshot_observed_at":"2026-08-06T21:52:23.674340Z","title":"Qwen2-vl: Enhancing vision-language model’s perception of the world at any resolution","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:23.674340Z"},"links":{"cited_paper":"/paper/2409.12191","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:29a871ea0639d712969b58a7a8515d43138dc89c362fb97e4a0a5471d29bb253","observation_id":"985fb899-f005-463c-b1e5-e3499e64c72f","resolution":{"observed_at":"2026-08-06T21:52:23.674340Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08394","last_updated":"2024-12-31T05:35:05Z","snapshot_observed_at":"2026-08-12T23:44:22.154771Z","submitted_at":"2024-06-12T16:44:50Z","title":"VisionLLM v2: An End-to-End Generalist Multimodal Large Language Model for Hundreds of Vision-Language Tasks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08394","snapshot_observed_at":"2026-08-06T21:52:23.743835Z","title":"Visionllm v2: An end-to-end general- ist multimodal large language model for hundreds of vision- language tasks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:23.743835Z"},"links":{"cited_paper":"/paper/2406.08394","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:b25f1a2b37e48cf007268603224a588e594668ad42dc815bc2c71940133f6874","observation_id":"bf746aaa-b2a3-428e-94c4-d7883a7ee9c8","resolution":{"observed_at":"2026-08-06T21:52:23.743835Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:30.873617Z","title":"V*: Guided visual search as a core mechanism in multimodal llms","venue":null,"work_id":"f0379d4a-5d63-4a48-9397-e8b1699ea968","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:23.852374Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:a3178a93dd2ee4add4e4a6a288f7e8ea00ae71e4faab39c6821056934bf87497","observation_id":"ed1fafd1-c4d0-420a-9d24-a1223f0adc7b","resolution":{"observed_at":"2026-08-06T21:52:30.950965Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:30.739046Z","title":"RealworldQA Dataset","venue":null,"work_id":"eb565141-2831-46ba-899b-0adb5a4cc1b7","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:23.943108Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:c3aedd097277f9268edd875839301c9a236e9df9adebb5e853dbf7d392c4a26c","observation_id":"3f187676-0180-47b0-941b-b631ccba57d8","resolution":{"observed_at":"2026-08-06T21:52:30.803340Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:30.577454Z","title":"Analyz- ing large language models’ capability in location prediction","venue":null,"work_id":"591ebf3c-7735-49b9-bf7e-1f8e28158402","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:24.066125Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:0eccf6f7f1299821efab6570ac9f0d3466f1e5c471795bd8d6301c805623285f","observation_id":"83b5e54f-4f66-4e79-a575-7d0fce68ee2b","resolution":{"observed_at":"2026-08-06T21:52:30.663944Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11813","last_updated":"2023-12-19T03:12:13Z","snapshot_observed_at":"2026-08-13T04:58:58.566957Z","submitted_at":"2023-12-19T03:12:13Z","title":"Urban Generative Intelligence (UGI): A Foundational Platform for Agents in Embodied City Environment","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11813","snapshot_observed_at":"2026-08-06T21:52:24.182301Z","title":"Ur- ban generative intelligence (ugi): A foundational platform for agents in embodied city environment","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:24.182301Z"},"links":{"cited_paper":"/paper/2312.11813","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:630a14eae9152592f90e62e3e84f1ce89af74b7432178541e9745e389c970ca0","observation_id":"8dc2dfb8-1c7c-4968-8c4e-506a1669c93f","resolution":{"observed_at":"2026-08-06T21:52:24.182301Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09686","last_updated":"2025-01-23T08:44:44Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-16T17:37:58Z","title":"Towards Large Reasoning Models: A Survey of Reinforced Reasoning with Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.09686","snapshot_observed_at":"2026-08-06T21:52:24.272058Z","title":"Towards large rea- soning models: A survey of reinforced reasoning with large language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:24.272058Z"},"links":{"cited_paper":"/paper/2501.09686","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:d4ad1b0e243ae1476bb2899b4f89914d1caeb2f7cab9961d3bcf4088b40e559a","observation_id":"9df47446-7336-44b2-a638-89adaf931ff2","resolution":{"observed_at":"2026-08-06T21:52:24.272058Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:30.396042Z","title":"Par- ticipatory cultural mapping based on collective behavior data in location-based social networks","venue":null,"work_id":"3696a1ca-4930-4ae9-8acc-0a28cd0bf861","year":null},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:24.356229Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:ef5a5f7448da007d87ead79e266a6ac54610b69912deed1b5295bc24211e54b7","observation_id":"a42cc69e-bb69-44fa-b228-7a09bb0316a3","resolution":{"observed_at":"2026-08-06T21:52:30.479881Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.13549","last_updated":"2024-11-29T15:51:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-23T15:21:52Z","title":"A Survey on Multimodal Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.13549","snapshot_observed_at":"2026-08-06T21:52:24.419530Z","title":"A survey on multimodal large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:24.419530Z"},"links":{"cited_paper":"/paper/2306.13549","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:933b721fe56a0e79158c40d4d8b5296bb9f4fb10dccdabe619b5d19770b21e26","observation_id":"af2619d3-6b9d-4e69-a7fa-59a751aa8b8f","resolution":{"observed_at":"2026-08-06T21:52:24.419530Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:30.217616Z","title":"Mm-vet: Evaluating large multimodal models for inte- grated capabilities","venue":null,"work_id":"cc8e35d6-0875-413c-8f2f-4c05f8f8c281","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:24.484165Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:52641a99e138d8984d4dbe1ae9b8027c442dcc5749e5a0529705e62e3aed11b7","observation_id":"cd0b7066-f87c-4c05-adf0-3d98eb6d0e1d","resolution":{"observed_at":"2026-08-06T21:52:30.296895Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.09712","last_updated":"2024-01-18T04:10:20Z","snapshot_observed_at":"2026-08-15T10:05:37.491633Z","submitted_at":"2024-01-18T04:10:20Z","title":"SkyEyeGPT: Unifying Remote Sensing Vision-Language Tasks via Instruction Tuning with Large Language Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.09712","snapshot_observed_at":"2026-08-06T21:52:24.557582Z","title":"Skyeyegpt: Unifying remote sensing vision-language tasks via instruc- tion tuning with large language model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:24.557582Z"},"links":{"cited_paper":"/paper/2401.09712","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:bd2a44b28af4eb55a8357dd463dbe1fb45688311979c89ae1f512e06391363f8","observation_id":"4ea88e8a-8db0-4ce1-8d73-77ef38fac776","resolution":{"observed_at":"2026-08-06T21:52:24.557582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:30.029521Z","title":"Earthgpt: A universal multi-modal large lan- guage model for multi-sensor image comprehension in re- mote sensing domain","venue":null,"work_id":"efce21b3-6278-4ee5-829c-6a869f142c18","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:24.628535Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:1ee9394d750ea5ab8198d9c7bdacb275710a905db8b2ac82703c3909bac74bb0","observation_id":"fb0d3055-79a3-4d39-af6d-94273c5476cc","resolution":{"observed_at":"2026-08-06T21:52:30.120055Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:29.897486Z","title":"Urban foundation models: A survey","venue":null,"work_id":"e698ec71-25b3-4844-a258-2e705472c7f0","year":null},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:24.672320Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:0fe0ea3648ec4e962dde98833b1fdac3faa54ea9f3f83b466eab0a2ab59f2a14","observation_id":"88ad8c4c-0f1e-420f-8a82-ee386e1cc1c4","resolution":{"observed_at":"2026-08-06T21:52:29.951040Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:29.738356Z","title":"UrbanMLLM: Joint learning of cross-view imagery for urban understanding, 2025","venue":null,"work_id":"9f1bd4a4-e91f-4d4a-b925-d42990e03bc2","year":2025},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:24.721919Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:bb8e4e9c9e753dffcc224feee120dcc3f6fca2c664c4a82da10b88449b4e5822","observation_id":"f16343f6-e71c-4829-b6df-6ff60778e59a","resolution":{"observed_at":"2026-08-06T21:52:29.815763Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:29.536403Z","title":"Per- ceiving urban inequality from imagery using visual language models with chain-of-thought reasoning","venue":null,"work_id":"b8f9f4f6-e9ae-435d-8d26-84b21cd39c3e","year":2025},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:24.801393Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:0eecfb06dc8da060e9c8aabbf7bff5ffe2977a60d25960c04f161bd0361fa77f","observation_id":"5f8b831f-e89b-4263-a0c4-c25564bd2048","resolution":{"observed_at":"2026-08-06T21:52:29.618477Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:29.357883Z","title":"Urbench: A comprehensive bench- mark for evaluating large multimodal models in multi-view urban scenarios","venue":null,"work_id":"6b73fee9-3b9d-467a-bd40-1145521b1f58","year":2025},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:24.854721Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:58b2f468d2c0f72a9b312e1b2ec17496dcf9cdc33dd335a630d7d4b20b4a5961","observation_id":"d2bfc31c-65af-4250-886b-738b385c421d","resolution":{"observed_at":"2026-08-06T21:52:29.440727Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:29.167299Z","title":"Deep learning for cross-domain data fu- sion in urban computing: Taxonomy, advances, and outlook","venue":null,"work_id":"fd3725aa-b8aa-4f87-8b6f-e7ac9ba25c00","year":2025},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:24.894192Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:a08232004391a69c82e2a24505d3f8366a45a23e1deff5f64b42b289c508f154","observation_id":"29ccec06-f7c7-4493-9da2-98a5f2a90a87","resolution":{"observed_at":"2026-08-06T21:52:29.257763Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:28.889863Z","title":"Figure 9","venue":null,"work_id":"fd4cad17-6bf1-4599-9b9c-4b53e92ea55e","year":null},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:24.963116Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:58412909f87497690d0c94e026974ff7b68e99ed38b19249ae1e10959cf369f9","observation_id":"ef3ed754-043d-4ed6-bdfa-550a9d2985bb","resolution":{"observed_at":"2026-08-06T21:52:29.008529Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:28.620238Z","title":null,"venue":null,"work_id":"1531348f-0d09-48ce-bb8b-f5482f6cef6b","year":null},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:25.019923Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:4f4f669886793c500bc00be69f1f6be369bf9effccdd61047f1a3bfd57f8b22c","observation_id":"f83a9c12-7d2c-48a3-a27f-632c538467da","resolution":{"observed_at":"2026-08-06T21:52:28.721388Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:28.408659Z","title":"Table 2 in Section 3.2 is the aggregated results of these three tables","venue":null,"work_id":"dafabbb8-ad15-49d2-a325-7ac807f9a564","year":null},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:25.067667Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:c7805f67e1e4f3f3a8206f79a0fd82b73834954503669a25ba371aa81d571fb9","observation_id":"69414709-d8dc-4be6-ad26-cff32443ebda","resolution":{"observed_at":"2026-08-06T21:52:28.498738Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:28.115309Z","title":null,"venue":null,"work_id":"26dc772d-bb54-4073-a2a5-8798df7204a8","year":null},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:25.119209Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:130e025f2a28df5f3b3f3849fb1d0cc0b528170fcbe7af88bd87a40f85ac2064","observation_id":"422b2167-4930-4227-b49b-cd6c734d3fad","resolution":{"observed_at":"2026-08-06T21:52:28.278570Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:27.832901Z","title":"11 presents training results with different amounts, ex- hibiting the high quality of UData","venue":null,"work_id":"67b65741-a8aa-4935-86a9-e0a8e41cf41b","year":null},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:25.203938Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:22fb01f01a9a02df87f9619df631b28d07a3f032d2bbaca04ebffc2a49d2b527","observation_id":"fd295feb-c4b9-4d64-964b-22449050b58c","resolution":{"observed_at":"2026-08-06T21:52:27.972132Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:27.483020Z","title":null,"venue":null,"work_id":"8122fa6b-eae9-43fe-b49d-da66d5ffb0b2","year":null},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:25.253531Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:dea8b2112b40de2a616f87ea7b6848302062d8ab13f972dabc3e593b8ec1fb91","observation_id":"8344fae6-6c4b-4b8f-a970-3f0b5373f535","resolution":{"observed_at":"2026-08-06T21:52:27.625332Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:27.250148Z","title":"However, for certain tasks, models of different sizes exhibit similar capabilities","venue":null,"work_id":"7819c4ce-de75-4c0f-9db6-9eb0c33b57ec","year":2000},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:25.315959Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:f417e5c0d4567442d75fded9eaa1364e088c44aed45a1b002c111dd7ea1abb9b","observation_id":"7385252b-4d0d-42e6-8b30-6911e05c794a","resolution":{"observed_at":"2026-08-06T21:52:27.351086Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:26.944858Z","title":"This task needs a model to speculate the land use type (commercial, residential, agricultural, etc.) based on a satellite image","venue":null,"work_id":"c3be549d-ca24-4d0a-abcb-f8fbb23c9da2","year":1920},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:25.382008Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:c02b1913bdea8f53af30d8c789434c4db25ca4e7839e44cacdc5cade5d20f931","observation_id":"bd8fd18b-f630-4512-a500-92c471c491ba","resolution":{"observed_at":"2026-08-06T21:52:27.097941Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:26.524227Z","title":null,"venue":null,"work_id":"79a0fccf-2176-4b31-aa7f-e40c0807cdbe","year":null},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:25.446679Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:4910a210073bf765a1d66bb7966e573c2c744d4c863039d985c87798e627d5a6","observation_id":"272f3bd7-3e1a-488c-b297-f5dd07e3e998","resolution":{"observed_at":"2026-08-06T21:52:26.762359Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-14T06:43:19.997848Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding"},"reference_resolution":{"displayed":66,"state_counts":{"malformed_identifier":2,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":28,"verified_exact":1,"verified_fuzzy":35},"total_outbound_references":66},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"thesis":"As of 15 August 2026, this Paper Citation Record lists 66 of 66 outbound references and 2 inbound Pith citation observations for arXiv:2506.23219."}