{"as_of":"2026-08-20T23:04:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:1f0ca3b7299ed071159f6fd139b62565c53d0e5f41f8e39deaf4af5529a73538","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":31,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":31,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-20T06:33:59.587034+00:00","state":"measured"},{"denominator":31,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":31,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T15:24:47.666879Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-10T09:37:00.841150Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2505.17012","last_updated":"2026-04-13T12:33:41Z","snapshot_observed_at":"2026-07-06T21:28:43.610849Z","submitted_at":"2025-05-22T17:59:03Z","title":"SpatialScore: Towards Comprehensive Evaluation for Spatial Intelligence","version":3},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-05-22T13:07:11.548885Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2505.17012"},"observation_digest":"sha256:029a6185f28cdc216ed25012019b8bef4bcb387b15f232ccfebab4478ebc4911","observation_id":"e770b902-a529-42f7-91e2-4410b4695509","resolution":{"observed_at":"2026-05-22T13:11:35.751741Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-08-07T13:50:33.010530Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.20728","last_updated":"2025-08-26T03:25:38Z","snapshot_observed_at":"2026-08-19T23:44:25.543802Z","submitted_at":"2025-05-27T05:17:41Z","title":"Jigsaw-Puzzles: From Seeing to Understanding to Reasoning in Vision-Language Models","version":4},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-07T13:50:33.010530Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2505.20728"},"observation_digest":"sha256:939a7b8e70fc7426dcd655115d628e6b6d31ad7cd8ab74894a00610ca73b5210","observation_id":"fc5d705f-b0e8-4de9-a506-d73312103c29","resolution":{"observed_at":"2026-08-07T13:50:33.010530Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-08-07T10:35:43.691323Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.05440","last_updated":"2025-06-05T12:43:10Z","snapshot_observed_at":"2026-08-07T21:22:23.284742Z","submitted_at":"2025-06-05T12:43:10Z","title":"BYO-Eval: Build Your Own Dataset for Fine-Grained Visual Assessment of Multimodal Language Models","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T10:35:43.691323Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2506.05440"},"observation_digest":"sha256:1ba82b36724834f1078a097d240e938950efa596755a69ac9950e2f38876a75d","observation_id":"0cd3df1a-b09a-4cb4-b3a5-a6eaa402bac7","resolution":{"observed_at":"2026-08-07T10:35:43.691323Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-08-07T00:57:27.247850Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-16T05:58:22.870564Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:27.247850Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:921f1c49aa2d5c9bbb057cba9b931aebc1b93fa81fb439f38043719a4ccf649b","observation_id":"6e20543d-3d2d-4381-8c77-b14dd9b70932","resolution":{"observed_at":"2026-08-07T00:57:27.247850Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-08-05T15:17:54.990372Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.20068","last_updated":"2025-08-27T17:22:34Z","snapshot_observed_at":"2026-08-13T23:17:38.070295Z","submitted_at":"2025-08-27T17:22:34Z","title":"11Plus-Bench: Demystifying Multimodal LLM Spatial Reasoning with Cognitive-Inspired Analysis","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-05T15:17:54.990372Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2508.20068"},"observation_digest":"sha256:d5a7a22d9dc1bc88514e20e2681f28c718f9a1cf652f496c44c0205b00e7fc57","observation_id":"f1ff9a6f-f2be-416f-a1c5-4836384f0f1a","resolution":{"observed_at":"2026-08-05T15:17:54.990372Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-08-03T16:55:22.331240Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2512.11393","last_updated":"2026-08-13T11:40:44Z","snapshot_observed_at":"2026-08-16T23:11:50.762075Z","submitted_at":"2025-12-12T09:07:21Z","title":"The N-Body Problem: Parallel Execution from Single-Person Egocentric Video","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-03T16:55:22.331240Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2512.11393"},"observation_digest":"sha256:65dd7e32aec7641a5eaedcade3c30dfb6d280879818dfe1624f61c2b1f1d707c","observation_id":"e2578177-12ce-4060-8b52-e60f60d175a6","resolution":{"observed_at":"2026-08-03T16:55:22.331240Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-08-03T16:30:36.756028Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2512.13517","last_updated":"2026-05-28T15:09:34Z","snapshot_observed_at":"2026-08-14T14:17:46.093839Z","submitted_at":"2025-12-15T16:43:50Z","title":"A Deep Learning Model of Mental Rotation Informed by Interactive VR Experiments","version":2},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-03T16:30:36.756028Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2512.13517"},"observation_digest":"sha256:713aced1309a265489e5d55b6bec076e8655f8f8a5d039da215ab31724e7fe63","observation_id":"7b3245e6-35e9-49b4-9611-369e9c34a280","resolution":{"observed_at":"2026-08-03T16:30:36.756028Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2602.11635","last_updated":"2026-04-08T16:46:28Z","snapshot_observed_at":"2026-08-20T04:34:30.705568Z","submitted_at":"2026-02-12T06:37:55Z","title":"Do MLLMs Really Understand Space? A Mathematical Reasoning Evaluation","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-16T03:39:28.364183Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2602.11635"},"observation_digest":"sha256:4f306bdb6c4eff0ca094acaa9253ac2c2eb13710de0eed7b9891152dae9933d2","observation_id":"f20761b5-2bc6-43be-86e5-777dedd3d6af","resolution":{"observed_at":"2026-05-16T03:40:33.119630Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-15T13:03:40.459461Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.08011","last_updated":"2026-05-23T08:33:03Z","snapshot_observed_at":"2026-08-17T13:06:45.716705Z","submitted_at":"2026-03-09T06:33:49Z","title":"It's Time to Get It Right: Improving Analog Clock Reading and Clock-Hand Spatial Reasoning in Vision-Language Models","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-07-15T13:03:40.459461Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2603.08011"},"observation_digest":"sha256:e0a19664114cc70f8ab3065af659c40f621599bf82b652a89e9afa3760207914","observation_id":"35b1f545-9ce4-4c8f-8068-6e79edf83bff","resolution":{"observed_at":"2026-07-15T13:03:40.459461Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2604.15184","last_updated":"2026-04-27T16:57:52Z","snapshot_observed_at":"2026-08-10T22:05:32.806329Z","submitted_at":"2026-04-16T16:15:23Z","title":"Agent-Aided Design for Dynamic CAD Models","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-10T11:13:01.515933Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2604.15184"},"observation_digest":"sha256:d46440f9182d35391b51a6378417f2fbcc3db9f8ef1a911e1ab7f9f1647aeada","observation_id":"684a2e35-f9b4-4673-823d-6c853d42bafb","resolution":{"observed_at":"2026-05-10T11:15:10.441699Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2604.16022","last_updated":"2026-04-17T12:51:46Z","snapshot_observed_at":"2026-08-13T02:15:13.699157Z","submitted_at":"2026-04-17T12:51:46Z","title":"SocialGrid: A Benchmark for Planning and Social Reasoning in Embodied Multi-Agent Systems","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-05-10T08:45:54.303143Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2604.16022"},"observation_digest":"sha256:2aebba988eda41aa297c98f88e2ca6f3e27c58f590cd166020da63ed8b5e02a4","observation_id":"276ae1ac-3d73-47c1-8eeb-35a9d0672089","resolution":{"observed_at":"2026-05-10T08:48:01.663869Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2604.26614","last_updated":"2026-04-29T12:41:39Z","snapshot_observed_at":"2026-08-11T13:51:24.959545Z","submitted_at":"2026-04-29T12:41:39Z","title":"State Beyond Appearance: Diagnosing and Improving State Consistency in Dial-Based Measurement Reading","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-07T11:45:57.291112Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2604.26614"},"observation_digest":"sha256:61ad2231d13dc46cd2610ebe521b0d4e8713833aadf9ac10e04a9d9285b3373c","observation_id":"958a3f9d-f2a8-4c50-9831-c97aa925b0de","resolution":{"observed_at":"2026-05-12T09:16:26.452568Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2605.09449","last_updated":"2026-05-10T10:01:57Z","snapshot_observed_at":"2026-08-14T01:42:23.694725Z","submitted_at":"2026-05-10T10:01:57Z","title":"SpaceMind++: Toward Allocentric Cognitive Maps for Spatially Grounded Video MLLMs","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-12T04:43:51.900721Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2605.09449"},"observation_digest":"sha256:ca729610869d024a6be05182caec70401331fb646590d06edd05c37745a7cc7d","observation_id":"2d20f7e4-2d5f-43a6-9df0-955f61c6ee48","resolution":{"observed_at":"2026-05-12T05:56:45.651024Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2605.09693","last_updated":"2026-05-10T18:25:52Z","snapshot_observed_at":"2026-08-11T14:36:51.006168Z","submitted_at":"2026-05-10T18:25:52Z","title":"Do multimodal models imagine electric sheep?","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-12T03:24:01.933339Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2605.09693"},"observation_digest":"sha256:5def794c611242ccc65c4c9f0a3722842758e720d0740d982d685e5362c21a24","observation_id":"8263d4c3-7d96-4686-bf6a-5eedab9bdb88","resolution":{"observed_at":"2026-05-12T03:26:19.535821Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2605.13169","last_updated":"2026-05-15T16:50:42Z","snapshot_observed_at":"2026-08-11T16:40:12.299909Z","submitted_at":"2026-05-13T08:31:22Z","title":"PanoWorld: Towards Spatial Supersensing in 360$^\\circ$ Panorama World","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-14T20:40:59.877854Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2605.13169"},"observation_digest":"sha256:9a93699c391d2d8a430396e1ec44756e846f860b0a83d46a17c399bb0ff4c6ca","observation_id":"82189804-e2b1-4d7a-9bb5-05fbf4032280","resolution":{"observed_at":"2026-05-14T20:42:57.648243Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2605.13169","last_updated":"2026-05-15T16:50:42Z","snapshot_observed_at":"2026-08-11T16:40:12.299909Z","submitted_at":"2026-05-13T08:31:22Z","title":"PanoWorld: Towards Spatial Supersensing in 360$^\\circ$ Panorama World","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-19T16:57:03.172340Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2605.13169"},"observation_digest":"sha256:28f05d44d6c38e8b09f07e29f527a2b22cebf2cbde3bd670da2eae55eb4bbd8e","observation_id":"4008e5b3-43d7-4822-adda-ec54db1b1db0","resolution":{"observed_at":"2026-05-19T16:57:40.176444Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2605.20448","last_updated":"2026-06-18T10:11:04Z","snapshot_observed_at":"2026-08-19T14:13:56.576927Z","submitted_at":"2026-05-19T20:01:19Z","title":"Do Vision-Language Models Understand 3D Scenes or Just Catalogue Objects?","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-21T06:55:04.657347Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2605.20448"},"observation_digest":"sha256:ef923abcb4055ecff1effdb5d88d989aa8684d7980b8173cfe7c18dd601fbb8e","observation_id":"ebe4fb80-1f1c-49cd-835c-d0a57767b29e","resolution":{"observed_at":"2026-05-21T06:59:45.742535Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2605.20448","last_updated":"2026-06-18T10:11:04Z","snapshot_observed_at":"2026-08-19T14:13:56.576927Z","submitted_at":"2026-05-19T20:01:19Z","title":"Do Vision-Language Models Understand 3D Scenes or Just Catalogue Objects?","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-06-30T17:52:26.785086Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2605.20448"},"observation_digest":"sha256:38c57529cc028539aa6bf6eff5ef609f1de599094aa3cb1095188e6e5596a691","observation_id":"64dbe084-5d50-408c-acab-981e84e68547","resolution":{"observed_at":"2026-06-30T17:54:57.791672Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2605.23141","last_updated":"2026-05-22T01:43:32Z","snapshot_observed_at":"2026-08-18T17:14:48.084695Z","submitted_at":"2026-05-22T01:43:32Z","title":"VisAnalog: A Diagnostic Suite for Visual Concept Transfer on Natural Images","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-25T05:17:45.034348Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2605.23141"},"observation_digest":"sha256:e98f31de2d785f14bc03252e637d6048b36be08e9a3bf1dd8f43ac9c27024896","observation_id":"2730b754-ea6a-4b0e-8042-3c4808aaee64","resolution":{"observed_at":"2026-05-25T05:20:24.981533Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2605.23176","last_updated":"2026-06-15T18:43:33Z","snapshot_observed_at":"2026-08-18T11:09:47.441160Z","submitted_at":"2026-05-22T02:52:06Z","title":"DRIVESPATIAL: A Benchmark for Spatiotemporal Intelligence in VLMs for Autonomous Driving","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-25T05:10:32.522453Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2605.23176"},"observation_digest":"sha256:eddd67f26410cb1f81212f85f0e16a26d5085f38426c5d1c3a66775548de0677","observation_id":"7577280a-8cca-43d9-a91f-8c3dab8ac721","resolution":{"observed_at":"2026-05-25T05:15:22.790290Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2605.23176","last_updated":"2026-06-15T18:43:33Z","snapshot_observed_at":"2026-08-18T11:09:47.441160Z","submitted_at":"2026-05-22T02:52:06Z","title":"DRIVESPATIAL: A Benchmark for Spatiotemporal Intelligence in VLMs for Autonomous Driving","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-06-30T16:40:22.441025Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2605.23176"},"observation_digest":"sha256:ba1ba58ef613bc00128d5ea28ab7698708ebe3f75146b8b4c1a91a7d54744f0e","observation_id":"d18c83e4-783f-4d0f-898b-e86c4781003e","resolution":{"observed_at":"2026-06-30T16:44:56.045464Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2605.23771","last_updated":"2026-05-22T15:40:52Z","snapshot_observed_at":"2026-08-14T17:01:43.585333Z","submitted_at":"2026-05-22T15:40:52Z","title":"PhotoFlow: Agentic 3D Virtual Photography Missions","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-25T04:33:24.355622Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2605.23771"},"observation_digest":"sha256:a71c2463a71d9f9d4ae81f8624f8c0f2f3afb09cb15f77626c3f90a6f095ed69","observation_id":"c0bcbcf9-3c93-4616-9b86-1034cf286b57","resolution":{"observed_at":"2026-05-25T04:35:21.022926Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2605.30557","last_updated":"2026-05-28T20:44:47Z","snapshot_observed_at":"2026-08-15T13:58:33.370405Z","submitted_at":"2026-05-28T20:44:47Z","title":"Seeing Isn't Knowing: Do VLMs Know When Not to Answer Spatial Questions (and Why)?","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-29T07:48:19.295578Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2605.30557"},"observation_digest":"sha256:459563c076b2b71ae65adae178bab0fc35491148fa041798de41f792dd3332cc","observation_id":"27886b22-3ddf-456c-9d55-27e77cd0e545","resolution":{"observed_at":"2026-06-29T07:53:13.577911Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2606.07641","last_updated":"2026-06-01T14:16:09Z","snapshot_observed_at":"2026-08-11T19:58:18.830064Z","submitted_at":"2026-06-01T14:16:09Z","title":"Readable Yet Unpredictable: Rotated-Outcome Prediction in Vision-Language Models","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-28T15:21:16.342973Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2606.07641"},"observation_digest":"sha256:de434ae4cc8b431ac635a04716750d484516a93429b2b12db6a3c4f55fd4a4b1","observation_id":"63fd505f-af17-4c8f-b2e4-3dc7eb652d1b","resolution":{"observed_at":"2026-07-01T22:26:18.105066Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2606.11918","last_updated":"2026-06-17T09:46:25Z","snapshot_observed_at":"2026-08-15T04:29:50.319316Z","submitted_at":"2026-06-10T10:50:06Z","title":"The Art of Interrogation: Consistency Amplifies Factuality in Spatial Reasoning","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-06-27T09:46:16.088490Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2606.11918"},"observation_digest":"sha256:911a2002b46c8cf1d4b8f02148b9ce2197380cb4f21aceab7af44f1bc0a66aef","observation_id":"68fdc119-0b54-4194-b323-bcbc6f4668fc","resolution":{"observed_at":"2026-07-03T10:58:03.046297Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2606.30378","last_updated":"2026-06-29T14:38:20Z","snapshot_observed_at":"2026-08-17T11:16:23.738867Z","submitted_at":"2026-06-29T14:38:20Z","title":"OmniCoT: A Benchmark for Global and Multi-Step Panoramic Reasoning","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-30T06:11:09.693576Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2606.30378"},"observation_digest":"sha256:bf9bd1321da3807cbae3dc63c2da84d72996840bf95369e7d39f47e3d443f7e2","observation_id":"75c0083b-0814-4336-b217-bedf8ab6c357","resolution":{"observed_at":"2026-06-30T06:14:19.130332Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2606.31467","last_updated":"2026-06-30T10:46:23Z","snapshot_observed_at":"2026-07-07T00:05:12.546815Z","submitted_at":"2026-06-30T10:46:23Z","title":"AeroVerse-SatAgent: UAV-Satellite Collaborative Spatial Reasoning Inspired by the Dual Visual Pathway Theory of Cognitive Neuroscience","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-07-01T06:18:35.494240Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2606.31467"},"observation_digest":"sha256:201edd4a252e20db870cc77366e79b1a944d136c7788cef94f883ef248f30e95","observation_id":"706e41e4-1ee3-4f9c-85a8-d07ae0816756","resolution":{"observed_at":"2026-07-01T09:45:39.862470Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2607.08317","last_updated":"2026-07-09T09:56:50Z","snapshot_observed_at":"2026-08-07T21:53:44.151713Z","submitted_at":"2026-07-09T09:56:50Z","title":"Blind-Spots-Bench: Evaluating Blind Spots in Multimodal Models","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-07-10T09:30:46.564013Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2607.08317"},"observation_digest":"sha256:c762cc5e822a143981609909ca0d3c29f8769b491f5b68635b936590d795fe90","observation_id":"a6d702a7-d72f-4393-af86-8a1340ef24b8","resolution":{"observed_at":"2026-07-10T09:37:00.842655Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-08-01T17:25:10.666535Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17657","last_updated":"2026-07-20T08:07:22Z","snapshot_observed_at":"2026-08-17T23:04:03.272648Z","submitted_at":"2026-07-20T08:07:22Z","title":"OrientSAM: Mitigating Camera-Centric Shortcut in Multimodal Spatial Reasoning via Orientation-Aware Spatial Alignment","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-01T17:25:10.666535Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2607.17657"},"observation_digest":"sha256:ca6d6fbe489558971033511f4f10fcf2ac7b42517a9da052230309c07bf4f338","observation_id":"a2f61674-1d1c-4f5f-9868-4bde23c20bfe","resolution":{"observed_at":"2026-08-01T17:25:10.666535Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-08-01T08:39:39.509832Z","title":"Huajie Tan, Enshen Zhou, Zhiyu Li, Yijie Xu, Yuheng Ji, Xiansheng Chen, Cheng Chi, Pengwei Wang, Huizhu Jia, Yulong Ao, et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21072","last_updated":"2026-07-23T09:04:48Z","snapshot_observed_at":"2026-08-19T14:09:20.412548Z","submitted_at":"2026-07-23T09:04:48Z","title":"Show, Don't Tell: Evaluating Spatial Cognition in Generative Pixels Rather Than LLM Text","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-01T08:39:39.509832Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2607.21072"},"observation_digest":"sha256:9299b5c2430aa3e23583cf23a1d9adb336134ffbbe7956da9daeaf2eb9d05bbb","observation_id":"ce3e6dab-3e1f-4d8f-a056-01001d16d180","resolution":{"observed_at":"2026-08-01T08:39:39.509832Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-08-15T15:24:47.666879Z","title":"arXiv preprint arXiv:2503.19707 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00726","last_updated":"2026-08-01T15:47:52Z","snapshot_observed_at":"2026-08-17T22:49:17.630687Z","submitted_at":"2026-08-01T15:47:52Z","title":"Foveated Probes Recover Localized Binding Information in Vision Foundation Models","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-15T15:24:47.666879Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2608.00726"},"observation_digest":"sha256:cf5d79261ed9c31ac198974a467ecb41e11f66bde0331592abdc5d529443938d","observation_id":"591a34d0-2c22-4a3f-8869-c2f4ae3232e4","resolution":{"observed_at":"2026-08-15T15:24:47.666879Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2503.19707/citation-record","integrity":"/paper/2503.19707/integrity","json":"/paper/2503.19707/citation-record.json","paper":"/paper/2503.19707"},"outbound":[],"paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"thesis":"As of 20 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 31 inbound Pith citation observations for arXiv:2503.19707."}