{"as_of":"2026-08-09T12:48:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:c716e2a9c160d4bcb93e4d4971632adb3877730165c1021ea2e37bf256086391","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":36,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":36,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":36,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":36,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T13:38:02.242912Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T16:18:37.306944Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":"2406.13642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-07-03T16:18:37.306944Z","title":"Spatialbot: Precise spatial understanding with vision language models.arXiv preprint arXiv:2406.13642","venue":null,"work_id":"8038b5ef-8748-4dc8-870d-38bbab426489","year":2024},"citing_paper":{"arxiv_id":"2412.14171","last_updated":"2025-07-02T21:00:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-18T18:59:54Z","title":"Thinking in Space: How Multimodal Large Language Models See, Remember, and Recall Spaces","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-22T09:27:43.919941Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2412.14171"},"observation_digest":"sha256:c4d95ef3ce5b15425c5d48e018f6682ac2a827558f92ce5f844ea1eb5bb9b618","observation_id":"e251e3d7-f2cf-4e7b-a04c-63d90c83f091","resolution":{"observed_at":"2026-05-22T09:27:44.172737Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-08T13:38:02.242912Z","title":"arXiv preprint arXiv:2406.13642 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.07192","last_updated":"2025-02-11T02:32:32Z","snapshot_observed_at":"2026-08-08T13:29:16.664788Z","submitted_at":"2025-02-11T02:32:32Z","title":"OscNet: Machine Learning on CMOS Oscillator Networks","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-08T13:38:02.242912Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2502.07192"},"observation_digest":"sha256:b05ff9e42922103ba431dc48bca056724d552e33ae51b73b59e4394cae80e4f5","observation_id":"bbc29c12-0a68-4762-a83f-ceb18684cf1e","resolution":{"observed_at":"2026-08-08T13:38:02.242912Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-07T14:24:04.556539Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-07T15:04:36.637684Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:04.556539Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:cf3cda99fc898bbcfec7a33d688a29702ce29c4ac076f16208ce33571748f835","observation_id":"7b406e20-d445-4681-a415-e8c4673aa066","resolution":{"observed_at":"2026-08-07T14:24:04.556539Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-07T12:35:16.202407Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24257","last_updated":"2025-05-30T06:32:26Z","snapshot_observed_at":"2026-08-08T14:01:51.003077Z","submitted_at":"2025-05-30T06:32:26Z","title":"Out of Sight, Not Out of Context? Egocentric Spatial Reasoning in VLMs Across Disjoint Frames","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-07T12:35:16.202407Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2505.24257"},"observation_digest":"sha256:ed0b285cc5462a3b97341053b908902b887a80974659cbc935c302f07bc94a5d","observation_id":"a02f28ca-6b43-4393-b617-ad0ba9419286","resolution":{"observed_at":"2026-08-07T12:35:16.202407Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-07T00:51:24.318444Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12525","last_updated":"2025-06-14T14:52:38Z","snapshot_observed_at":"2026-08-07T14:31:47.265900Z","submitted_at":"2025-06-14T14:52:38Z","title":"A Spatial Relationship Aware Dataset for Robotics","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T00:51:24.318444Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2506.12525"},"observation_digest":"sha256:ca9c367400f265edaaff28fab612cdb4fab1d1459f631197632d762a03cd7125","observation_id":"e726471f-2f61-4b87-ae35-606d52eb03d2","resolution":{"observed_at":"2026-08-07T00:51:24.318444Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-07T00:51:48.968184Z","title":"Spatialbot: Precise spatial understanding with vision language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12610","last_updated":"2025-07-17T00:39:38Z","snapshot_observed_at":"2026-08-07T00:42:55.264522Z","submitted_at":"2025-06-14T19:10:23Z","title":"OscNet v1.5: Energy Efficient Hopfield Network on CMOS Oscillators for Image Classification","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T00:51:48.968184Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2506.12610"},"observation_digest":"sha256:c3e1f753b2041e155377ec9e541e72d377b87dabff624a32324291494b8f49b5","observation_id":"64ec50b4-7244-4a6f-b574-b2061b2a6d40","resolution":{"observed_at":"2026-08-07T00:51:48.968184Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-06T23:01:29.887022Z","title":"Spatialbot: Precise spatial understanding with vision lan- guage models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-08T00:48:32.373291Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.887022Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:273c46bb3d1ae0424613e1b65c71f9fdaa642987c5ef95bcfec96ef21f7dfa8d","observation_id":"ae73add2-f905-443d-b7aa-abbab6e06216","resolution":{"observed_at":"2026-08-06T23:01:29.887022Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-06T20:52:02.423508Z","title":"Spatialbot: Precise spatial understanding with vision language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.01634","last_updated":"2025-07-02T12:05:57Z","snapshot_observed_at":"2026-08-09T10:47:49.129397Z","submitted_at":"2025-07-02T12:05:57Z","title":"Depth Anything at Any Condition","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T20:52:02.423508Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2507.01634"},"observation_digest":"sha256:d3642f320e3bde92942297b39722a4102dc1a390ed5e03eef04753b7309b9684","observation_id":"27c51cb1-594a-4ea0-9033-28489f4484f4","resolution":{"observed_at":"2026-08-06T20:52:02.423508Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-06T21:22:27.079986Z","title":"Spatialbot: Precise spatial understanding with vision language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02978","last_updated":"2025-07-01T03:05:56Z","snapshot_observed_at":"2026-08-08T00:06:19.287707Z","submitted_at":"2025-07-01T03:05:56Z","title":"Ascending the Infinite Ladder: Benchmarking Spatial Deformation Reasoning in Vision-Language Models","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T21:22:27.079986Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2507.02978"},"observation_digest":"sha256:f6050bc6c374e9b7e88fc48fa5488dc339c641236d25f6bd15a8319529528464","observation_id":"5404da64-bf32-4d31-9561-f7a31b58e557","resolution":{"observed_at":"2026-08-06T21:22:27.079986Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-06T20:15:52.733114Z","title":"Spatialbot: Precise spatial understanding with vision language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.03483","last_updated":"2025-07-08T05:05:04Z","snapshot_observed_at":"2026-08-09T01:40:47.049405Z","submitted_at":"2025-07-04T11:20:09Z","title":"BMMR: A Large-Scale Bilingual Multimodal Multi-Discipline Reasoning Dataset","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T20:15:52.733114Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2507.03483"},"observation_digest":"sha256:6d3ce4d92f6c0d2adc2c676d0bcd5c6e456c1e7e73666a2c85b7265bfb06bd25","observation_id":"c569df7c-8152-4629-8373-70c74568eca3","resolution":{"observed_at":"2026-08-06T20:15:52.733114Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-06T20:04:38.185326Z","title":"Spatialbot: Precise spatial understanding with vision lan- guage models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.03930","last_updated":"2025-07-08T01:07:30Z","snapshot_observed_at":"2026-08-07T12:49:17.624180Z","submitted_at":"2025-07-05T07:29:37Z","title":"RwoR: Generating Robot Demonstrations from Human Hand Collection for Policy Learning without Robot","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T20:04:38.185326Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2507.03930"},"observation_digest":"sha256:c4fb6218a6bbd6a11ad9edff65516ee947436ebb2db13b0611a281f7ae045184","observation_id":"dbb6d299-169c-4016-a284-3201659f70f7","resolution":{"observed_at":"2026-08-06T20:04:38.185326Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-06T19:48:07.793262Z","title":"Spatialbot: Precise spatial understanding with vision language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.04633","last_updated":"2025-07-07T03:28:03Z","snapshot_observed_at":"2026-08-07T22:42:00.724406Z","submitted_at":"2025-07-07T03:28:03Z","title":"PRISM: Pointcloud Reintegrated Inference via Segmentation and Cross-attention for Manipulation","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T19:48:07.793262Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2507.04633"},"observation_digest":"sha256:2d29b1de6c4afc59a8104df6f8bd4a7e5f737cd6efe91bea068636004a7d03fe","observation_id":"79ba07b9-c5b4-4f00-afe7-84d27c3f0500","resolution":{"observed_at":"2026-08-06T19:48:07.793262Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-06T17:29:38.536084Z","title":"Spatialbot: Precise spatial understanding with vision language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10778","last_updated":"2025-08-14T03:48:03Z","snapshot_observed_at":"2026-08-08T00:48:22.779591Z","submitted_at":"2025-07-14T20:05:55Z","title":"Warehouse Spatial Question Answering with LLM Agent","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T17:29:38.536084Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2507.10778"},"observation_digest":"sha256:ead2995a036551f2d5ddebfd2856f27aec9f4734f51535e283ba87a61a00abc1","observation_id":"e2ed1f8c-2a8d-4720-84e0-5316589660b5","resolution":{"observed_at":"2026-08-06T17:29:38.536084Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-06T19:54:48.156627Z","title":"Spatialbot: Precise spatial understanding with vision language models,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.13362","last_updated":"2025-07-06T10:51:12Z","snapshot_observed_at":"2026-08-08T08:07:57.477546Z","submitted_at":"2025-07-06T10:51:12Z","title":"Enhancing Spatial Reasoning in Vision-Language Models via Chain-of-Thought Prompting and Reinforcement Learning","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T19:54:48.156627Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2507.13362"},"observation_digest":"sha256:ff100173833dcdff69dee98fb09a35d7180cc589209c396c6196d0ca279d28da","observation_id":"46a2075a-2a04-4750-af4a-544bd439c12c","resolution":{"observed_at":"2026-08-06T19:54:48.156627Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-06T15:38:59.749275Z","title":"Spatialbot: Precise spatial understanding with vision language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15321","last_updated":"2025-07-21T07:23:14Z","snapshot_observed_at":"2026-08-07T10:01:48.797206Z","submitted_at":"2025-07-21T07:23:14Z","title":"BenchDepth: Are We on the Right Way to Evaluate Depth Foundation Models?","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T15:38:59.749275Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2507.15321"},"observation_digest":"sha256:6092e3a3b1a03530f1f77fccc235c7d5abd3edfb1141b5d8be4418a5bdf82b40","observation_id":"5957f5e9-d878-47cd-9323-840f7158ce10","resolution":{"observed_at":"2026-08-06T15:38:59.749275Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-05T22:23:09.481651Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.07135","last_updated":"2025-08-10T01:15:37Z","snapshot_observed_at":"2026-08-09T08:28:13.984259Z","submitted_at":"2025-08-10T01:15:37Z","title":"Canvas3D: Empowering Precise Spatial Control for Image Generation with Constraints from a 3D Virtual Canvas","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-05T22:23:09.481651Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2508.07135"},"observation_digest":"sha256:1ab8d443bfc64d7d290849174d79b8bfd017bba1f4ed78c34d7594d2e9fc2b66","observation_id":"1cfeaa5f-2de8-42bf-b014-002bbc930982","resolution":{"observed_at":"2026-08-05T22:23:09.481651Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":"2406.13642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-07-03T16:18:37.306944Z","title":"Spatialbot: Precise spatial understanding with vision language models.arXiv preprint arXiv:2406.13642","venue":null,"work_id":"8038b5ef-8748-4dc8-870d-38bbab426489","year":2024},"citing_paper":{"arxiv_id":"2508.13998","last_updated":"2026-04-06T03:29:44Z","snapshot_observed_at":"2026-08-03T11:31:47.613184Z","submitted_at":"2025-08-19T16:50:01Z","title":"Embodied-R1: Reinforced Embodied Reasoning for General Robotic Manipulation","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-18T22:04:34.235731Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2508.13998"},"observation_digest":"sha256:6be9ee45b3c0f046628c9240110c0a6a0e5c3482801c6def2d1e591e29eea731","observation_id":"452ba9fc-3031-4cb0-aa31-3c394b634645","resolution":{"observed_at":"2026-05-18T22:06:51.720215Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-05T11:56:07.891723Z","title":"Spatialbot: Precise spatial understanding with vision language models, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.02175","last_updated":"2025-09-04T16:38:44Z","snapshot_observed_at":"2026-08-05T11:56:06.572610Z","submitted_at":"2025-09-02T10:32:58Z","title":"Understanding Space Is Rocket Science -- Only Top Reasoning Models Can Solve Spatial Understanding Tasks","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-05T11:56:07.891723Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2509.02175"},"observation_digest":"sha256:a3f042dd764dc88c15e4d5528c93c879186d0cf7b6f934e99bf30e8930a6071e","observation_id":"69047047-b5e5-45cd-8174-14a6659374bd","resolution":{"observed_at":"2026-08-05T11:56:07.891723Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-04T11:38:11.546062Z","title":"Spatialbot: Precise spatial understanding with vision language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2510.03896","last_updated":"2026-06-11T03:17:22Z","snapshot_observed_at":"2026-08-09T06:57:01.817389Z","submitted_at":"2025-10-04T18:33:27Z","title":"GAE: Unleashing Physical Potential of VLM with Generalizable Action Expert","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-04T11:38:11.546062Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2510.03896"},"observation_digest":"sha256:dc2251f3f538d9b8cf6769b780363d9ef64dbbfcff7b38131a28be4e98ca3a52","observation_id":"82cd167f-a1ad-452d-a6d8-9646c13e431c","resolution":{"observed_at":"2026-08-04T11:38:11.546062Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":"2406.13642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-07-03T16:18:37.306944Z","title":"Spatialbot: Precise spatial understanding with vision language models.arXiv preprint arXiv:2406.13642","venue":null,"work_id":"8038b5ef-8748-4dc8-870d-38bbab426489","year":2024},"citing_paper":{"arxiv_id":"2511.06754","last_updated":"2026-05-06T07:00:01Z","snapshot_observed_at":"2026-07-29T01:21:11.279990Z","submitted_at":"2025-11-10T06:33:44Z","title":"SlotVLA: Towards Modeling of Object-Relation Representations in Robotic Manipulation","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-18T00:22:41.611893Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2511.06754"},"observation_digest":"sha256:38ba9df81351ffec6669ce4559529a9a512467226faf9f55596f14477ebc909e","observation_id":"a73e99a7-7552-4f1a-9e21-59d69faf6d4b","resolution":{"observed_at":"2026-05-18T00:25:32.994831Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-03T23:08:49.001243Z","title":"Spatialbot: Precise spatial understanding with vision language models.arXiv preprint arXiv:2406.13642,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2511.07403","last_updated":"2026-07-02T19:21:24Z","snapshot_observed_at":"2026-08-07T07:35:15.256090Z","submitted_at":"2025-11-10T18:52:47Z","title":"SpatialThinker: Reinforcing Scene Graph-Grounded Spatial Reasoning via Dense Rewards","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-03T23:08:49.001243Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2511.07403"},"observation_digest":"sha256:ee009de9f98fe5eafae741dbe503d507e183395842e612ee812005486da68d42","observation_id":"c89204c0-a66f-433b-9146-90382ced24ee","resolution":{"observed_at":"2026-08-03T23:08:49.001243Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":"2406.13642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-07-03T16:18:37.306944Z","title":"Spatialbot: Precise spatial understanding with vision language models.arXiv preprint arXiv:2406.13642","venue":null,"work_id":"8038b5ef-8748-4dc8-870d-38bbab426489","year":2024},"citing_paper":{"arxiv_id":"2511.17411","last_updated":"2026-04-27T17:16:04Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-11-21T17:09:43Z","title":"SPEAR-1: Scaling Beyond Robot Demonstrations via 3D Understanding","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-17T20:21:12.375936Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2511.17411"},"observation_digest":"sha256:2bb55e601ded9639d34b8dac9c93f6761323de26a4e324c6f952014d88711af1","observation_id":"f4991661-264f-47e3-b6ef-96f837d1bf93","resolution":{"observed_at":"2026-05-17T20:22:04.620503Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-03T20:38:54.925300Z","title":"Spatialbot: Precise spatial understanding with vision language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2511.19119","last_updated":"2026-06-28T11:41:55Z","snapshot_observed_at":"2026-08-08T23:45:04.753103Z","submitted_at":"2025-11-24T13:49:17Z","title":"MonoSR: Open-Vocabulary Spatial Reasoning from Monocular Images","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-03T20:38:54.925300Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2511.19119"},"observation_digest":"sha256:41726f37a45189b1f43bd650ceffb3a03ac05a1b973af94cabda0df0573876f4","observation_id":"89fc6ba7-bae3-419e-9de8-09a4738a8d4a","resolution":{"observed_at":"2026-08-03T20:38:54.925300Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":"2406.13642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-07-03T16:18:37.306944Z","title":"Spatialbot: Precise spatial understanding with vision language models.arXiv preprint arXiv:2406.13642","venue":null,"work_id":"8038b5ef-8748-4dc8-870d-38bbab426489","year":2024},"citing_paper":{"arxiv_id":"2604.07592","last_updated":"2026-04-08T20:49:50Z","snapshot_observed_at":"2026-07-06T22:55:46.307881Z","submitted_at":"2026-04-08T20:49:50Z","title":"Spatio-Temporal Grounding of Large Language Models from Perception Streams","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T17:10:45.837684Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2604.07592"},"observation_digest":"sha256:fe20e4ee533d73253d41bcdd55fa2c26bd626c5e6fc36f7093351033136b3c6d","observation_id":"0a578d09-6ef1-47c0-acc8-fb7471255bc4","resolution":{"observed_at":"2026-05-11T07:26:02.678079Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":"2406.13642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-07-03T16:18:37.306944Z","title":"Spatialbot: Precise spatial understanding with vision language models.arXiv preprint arXiv:2406.13642","venue":null,"work_id":"8038b5ef-8748-4dc8-870d-38bbab426489","year":2024},"citing_paper":{"arxiv_id":"2604.20012","last_updated":"2026-04-21T21:40:58Z","snapshot_observed_at":"2026-07-06T23:06:31.110859Z","submitted_at":"2026-04-21T21:40:58Z","title":"EmbodiedMidtrain: Bridging the Gap between Vision-Language Models and Vision-Language-Action Models via Mid-training","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T02:16:08.687340Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2604.20012"},"observation_digest":"sha256:df98e0f2244208cb507012b8972b5fd094160d6f8c298ff357385538e5e42471","observation_id":"64bbe8e8-686b-4ab3-8c2a-8a9ecec42f57","resolution":{"observed_at":"2026-05-11T13:11:03.815073Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":"2406.13642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-07-03T16:18:37.306944Z","title":"Spatialbot: Precise spatial understanding with vision language models.arXiv preprint arXiv:2406.13642","venue":null,"work_id":"8038b5ef-8748-4dc8-870d-38bbab426489","year":2024},"citing_paper":{"arxiv_id":"2605.05997","last_updated":"2026-05-22T12:07:44Z","snapshot_observed_at":"2026-08-02T14:48:34.696358Z","submitted_at":"2026-05-07T10:48:46Z","title":"4DThinker: Thinking with 4D Imagery for Dynamic Spatial Understanding","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-08T14:20:08.404090Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2605.05997"},"observation_digest":"sha256:189373883497c7609038b1a53fd12c65fd3c63d79d9d1779cbf16f38911435b9","observation_id":"54209770-f05e-4e37-ba5e-3fc297cfd849","resolution":{"observed_at":"2026-05-11T18:41:11.692955Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":"2406.13642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-07-03T16:18:37.306944Z","title":"Spatialbot: Precise spatial understanding with vision language models.arXiv preprint arXiv:2406.13642","venue":null,"work_id":"8038b5ef-8748-4dc8-870d-38bbab426489","year":2024},"citing_paper":{"arxiv_id":"2605.05997","last_updated":"2026-05-22T12:07:44Z","snapshot_observed_at":"2026-08-02T14:48:34.696358Z","submitted_at":"2026-05-07T10:48:46Z","title":"4DThinker: Thinking with 4D Imagery for Dynamic Spatial Understanding","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-25T06:15:33.062980Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2605.05997"},"observation_digest":"sha256:1cbb7b8fa9b2c7b2f10c81414dcf469aa4da3be529de4a267d644c4a3af8c6e3","observation_id":"fa29e92c-af28-48a0-a4ed-1caaaf27aa2f","resolution":{"observed_at":"2026-05-25T06:16:39.992535Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":"2406.13642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-07-03T16:18:37.306944Z","title":"Spatialbot: Precise spatial understanding with vision language models.arXiv preprint arXiv:2406.13642","venue":null,"work_id":"8038b5ef-8748-4dc8-870d-38bbab426489","year":2024},"citing_paper":{"arxiv_id":"2605.10588","last_updated":"2026-05-11T13:59:09Z","snapshot_observed_at":"2026-07-06T23:22:32.858399Z","submitted_at":"2026-05-11T13:59:09Z","title":"Thinking with Novel Views: A Systematic Analysis of Generative-Augmented Spatial Intelligence","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-05-12T03:24:41.877312Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2605.10588"},"observation_digest":"sha256:dae3bc6ac25b553bad74f236c0fd3275aec027f0e7bb6c2387486211dcf23921","observation_id":"9aa7b8c5-3f88-49ab-bdd7-ffe4c9f2a117","resolution":{"observed_at":"2026-05-12T03:26:19.382386Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":"2406.13642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-07-03T16:18:37.306944Z","title":"Spatialbot: Precise spatial understanding with vision language models.arXiv preprint arXiv:2406.13642","venue":null,"work_id":"8038b5ef-8748-4dc8-870d-38bbab426489","year":2024},"citing_paper":{"arxiv_id":"2605.18746","last_updated":"2026-05-25T08:34:52Z","snapshot_observed_at":"2026-08-02T07:58:31.278661Z","submitted_at":"2026-05-18T17:59:02Z","title":"ESI-Bench: Towards Embodied Spatial Intelligence that Closes the Perception-Action Loop","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-20T10:52:22.778489Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2605.18746"},"observation_digest":"sha256:dfb2987e5c1272ea13d596e67ed3041c56dfd31f6b2312516dcacba28e47538f","observation_id":"88753f51-bdb3-4f4a-892d-3971c0915305","resolution":{"observed_at":"2026-05-20T10:53:13.222326Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":"2406.13642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-07-03T16:18:37.306944Z","title":"Spatialbot: Precise spatial understanding with vision language models.arXiv preprint arXiv:2406.13642","venue":null,"work_id":"8038b5ef-8748-4dc8-870d-38bbab426489","year":2024},"citing_paper":{"arxiv_id":"2605.18746","last_updated":"2026-05-25T08:34:52Z","snapshot_observed_at":"2026-08-02T07:58:31.278661Z","submitted_at":"2026-05-18T17:59:02Z","title":"ESI-Bench: Towards Embodied Spatial Intelligence that Closes the Perception-Action Loop","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-30T18:25:17.831116Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2605.18746"},"observation_digest":"sha256:adbff533fbf43a58300040dbe19ae8a38c2009d74999fb1d6960fdc60159b2f4","observation_id":"b0018adf-4f46-4670-9535-fda813036c2d","resolution":{"observed_at":"2026-07-01T15:05:47.182481Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":"2406.13642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-07-03T16:18:37.306944Z","title":"Spatialbot: Precise spatial understanding with vision language models.arXiv preprint arXiv:2406.13642","venue":null,"work_id":"8038b5ef-8748-4dc8-870d-38bbab426489","year":2024},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:346bbc4870885af8f89e6f882bee2bf125f5fe23ec9fcb81fe8484d2a66c8787","observation_id":"016b2ce0-c946-4469-add7-ec1b9b10e22c","resolution":{"observed_at":"2026-06-29T22:13:59.560592Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":"2406.13642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-07-03T16:18:37.306944Z","title":"Spatialbot: Precise spatial understanding with vision language models.arXiv preprint arXiv:2406.13642","venue":null,"work_id":"8038b5ef-8748-4dc8-870d-38bbab426489","year":2024},"citing_paper":{"arxiv_id":"2605.30561","last_updated":"2026-05-28T20:48:55Z","snapshot_observed_at":"2026-07-06T23:39:48.531480Z","submitted_at":"2026-05-28T20:48:55Z","title":"VLM3: Vision Language Models Are Native 3D Learners","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-29T07:45:31.978215Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2605.30561"},"observation_digest":"sha256:3a9f1d380398e5ec37f12ce0cb67ed3bdeabe5a7058a14d61cffcb00817d77f8","observation_id":"effb4d79-bfe3-4ed5-b356-8d1378a6ecba","resolution":{"observed_at":"2026-06-29T07:53:14.019914Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":"2406.13642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-07-03T16:18:37.306944Z","title":"Spatialbot: Precise spatial understanding with vision language models.arXiv preprint arXiv:2406.13642","venue":null,"work_id":"8038b5ef-8748-4dc8-870d-38bbab426489","year":2024},"citing_paper":{"arxiv_id":"2606.05445","last_updated":"2026-06-03T21:08:06Z","snapshot_observed_at":"2026-08-06T20:42:38.181639Z","submitted_at":"2026-06-03T21:08:06Z","title":"Brick-Composer: Using MLLMs for Assembly with Diverse Bricks","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-28T05:59:29.302038Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2606.05445"},"observation_digest":"sha256:58e591e447877ad4ef7767dd4a01640f8257968c7b76bf850aed2f368f34e9ba","observation_id":"6f97f28b-55ff-41db-b7ee-b719312b5659","resolution":{"observed_at":"2026-07-02T08:26:48.424339Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":"2406.13642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-07-03T16:18:37.306944Z","title":"Spatialbot: Precise spatial understanding with vision language models.arXiv preprint arXiv:2406.13642","venue":null,"work_id":"8038b5ef-8748-4dc8-870d-38bbab426489","year":2024},"citing_paper":{"arxiv_id":"2606.31257","last_updated":"2026-06-30T07:33:18Z","snapshot_observed_at":"2026-07-07T00:04:58.486133Z","submitted_at":"2026-06-30T07:33:18Z","title":"Decodable Is Not Grounded: A Vision-Ablation Arbiter for VLM Spatial Reasoning","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-01T06:23:00.372251Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2606.31257"},"observation_digest":"sha256:7dffa0e082a092c3e23be25c12f12031b969fbcf582e49036763b4295b48bc79","observation_id":"443ec8bd-ee16-464b-b4fb-1deb9e64acf5","resolution":{"observed_at":"2026-07-01T09:35:41.110274Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":"2406.13642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-07-03T16:18:37.306944Z","title":"Spatialbot: Precise spatial understanding with vision language models.arXiv preprint arXiv:2406.13642","venue":null,"work_id":"8038b5ef-8748-4dc8-870d-38bbab426489","year":2024},"citing_paper":{"arxiv_id":"2607.01784","last_updated":"2026-07-02T06:56:29Z","snapshot_observed_at":"2026-08-03T11:23:56.400674Z","submitted_at":"2026-07-02T06:56:29Z","title":"SpaceEra++: A Unified Framework Towards 3D Spatial Reasoning in Video","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-03T16:16:41.412451Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2607.01784"},"observation_digest":"sha256:6e961dbde3d074f08f062dcfd97b173ec4401cb6603062a46b029911d00ccd13","observation_id":"bb376c85-f6f5-470b-b01f-51d3b0834949","resolution":{"observed_at":"2026-07-03T16:18:37.308596Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-06T21:49:20.587662Z","title":"arXiv preprint arXiv:2406.13642 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04574","last_updated":"2026-08-05T08:04:26Z","snapshot_observed_at":"2026-08-08T23:49:15.665258Z","submitted_at":"2026-08-05T08:04:26Z","title":"When Memory Lies: An Empirical Study of Spatial Memory Staleness in VLM Agents","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-06T21:49:20.587662Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2608.04574"},"observation_digest":"sha256:b941b75f08d68f78234543f6bf390d1c96c6f08e8e04dfee16650aaeeca885f3","observation_id":"83309848-96fa-4915-a231-5899935daab2","resolution":{"observed_at":"2026-08-06T21:49:20.587662Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2406.13642/citation-record","integrity":"/paper/2406.13642/integrity","json":"/paper/2406.13642/citation-record.json","paper":"/paper/2406.13642"},"outbound":[],"paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","latest_version":7,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 36 inbound Pith citation observations for arXiv:2406.13642."}