{"as_of":"2026-08-14T09:39:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:d49873b330827173202d376620cfb036e7ad55d20ca83828fe0217cb37bbae7d","coverage":[{"denominator":49,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":49,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T11:40:25.147462Z","state":"measured"},{"denominator":57,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":57,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-14T06:32:32.682623+00:00","state":"measured"},{"denominator":8,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":8,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T06:33:30.837244Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T19:50:10.271272Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"cited_work":{"arxiv_id":"2509.02359","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2509.02359","snapshot_observed_at":"2026-07-04T19:50:10.271272Z","title":"Why do mllms struggle with spatial understanding? a systematic analysis from data to architecture","venue":null,"work_id":"0ff3d2c2-b3b3-41b0-8bb3-bae4d41de054","year":2025},"citing_paper":{"arxiv_id":"2603.03944","last_updated":"2026-04-03T20:11:12Z","snapshot_observed_at":"2026-07-06T22:47:46.409064Z","submitted_at":"2026-03-04T11:09:39Z","title":"SCP: Spatial Causal Prediction in Video","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-15T16:47:44.523606Z"},"links":{"cited_paper":"/paper/2509.02359","citing_paper":"/paper/2603.03944"},"observation_digest":"sha256:414468dae9c427adf901211fad24a517d2f30aa195068695e485edf29150e8e3","observation_id":"ca2531c2-ce95-4d7d-83cd-3b5390d70603","resolution":{"observed_at":"2026-05-15T16:50:11.305915Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"cited_work":{"arxiv_id":"2509.02359","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2509.02359","snapshot_observed_at":"2026-07-04T19:50:10.271272Z","title":"Why do mllms struggle with spatial understanding? a systematic analysis from data to architecture","venue":null,"work_id":"0ff3d2c2-b3b3-41b0-8bb3-bae4d41de054","year":2025},"citing_paper":{"arxiv_id":"2604.13321","last_updated":"2026-04-14T21:57:58Z","snapshot_observed_at":"2026-08-11T20:04:12.538612Z","submitted_at":"2026-04-14T21:57:58Z","title":"Why MLLMs Struggle to Determine Object Orientations","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-10T15:19:44.979076Z"},"links":{"cited_paper":"/paper/2509.02359","citing_paper":"/paper/2604.13321"},"observation_digest":"sha256:8ea625539dbe95dd4fb57ba21490c5cf1d7b718ac572f60143b6a73c8c80e87c","observation_id":"f8784c33-9dd8-429b-956d-07c8ea5c2d13","resolution":{"observed_at":"2026-05-11T10:46:04.597092Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"cited_work":{"arxiv_id":"2509.02359","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2509.02359","snapshot_observed_at":"2026-07-04T19:50:10.271272Z","title":"Why do mllms struggle with spatial understanding? a systematic analysis from data to architecture","venue":null,"work_id":"0ff3d2c2-b3b3-41b0-8bb3-bae4d41de054","year":2025},"citing_paper":{"arxiv_id":"2605.22100","last_updated":"2026-05-28T08:19:59Z","snapshot_observed_at":"2026-08-14T08:44:29.115056Z","submitted_at":"2026-05-21T07:36:41Z","title":"MPDocBench-Parse: Benchmarking Practical Multi-page Document Parsing","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-22T05:58:04.055855Z"},"links":{"cited_paper":"/paper/2509.02359","citing_paper":"/paper/2605.22100"},"observation_digest":"sha256:225bd594ee1a51370ae7a0f7e2b514859f7ab40860a1d8dd9c788152d4ae8175","observation_id":"6be88569-d713-4f35-b426-8593bf50f9a3","resolution":{"observed_at":"2026-05-22T06:01:08.940684Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"cited_work":{"arxiv_id":"2509.02359","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2509.02359","snapshot_observed_at":"2026-07-04T19:50:10.271272Z","title":"Why do mllms struggle with spatial understanding? a systematic analysis from data to architecture","venue":null,"work_id":"0ff3d2c2-b3b3-41b0-8bb3-bae4d41de054","year":2025},"citing_paper":{"arxiv_id":"2605.22100","last_updated":"2026-05-28T08:19:59Z","snapshot_observed_at":"2026-08-14T08:44:29.115056Z","submitted_at":"2026-05-21T07:36:41Z","title":"MPDocBench-Parse: Benchmarking Practical Multi-page Document Parsing","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-06-30T17:37:33.750306Z"},"links":{"cited_paper":"/paper/2509.02359","citing_paper":"/paper/2605.22100"},"observation_digest":"sha256:5f6ed198253c120d7e0417cf8ce14a25cddf711fed629b89ff009c843e177e0c","observation_id":"9ee60d47-bac9-4064-951c-b7ec93060950","resolution":{"observed_at":"2026-07-01T15:05:48.241818Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"cited_work":{"arxiv_id":"2509.02359","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2509.02359","snapshot_observed_at":"2026-07-04T19:50:10.271272Z","title":"Why do mllms struggle with spatial understanding? a systematic analysis from data to architecture","venue":null,"work_id":"0ff3d2c2-b3b3-41b0-8bb3-bae4d41de054","year":2025},"citing_paper":{"arxiv_id":"2605.30161","last_updated":"2026-05-28T16:18:01Z","snapshot_observed_at":"2026-07-06T23:39:29.458653Z","submitted_at":"2026-05-28T16:18:01Z","title":"Why Far Looks Up: Probing Spatial Representation in Vision-Language Models","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-06-29T08:06:32.403727Z"},"links":{"cited_paper":"/paper/2509.02359","citing_paper":"/paper/2605.30161"},"observation_digest":"sha256:16bc2f160d763d819b332c5d7f90eaa5134895bb3b7f38dac4d3e5f892497ec7","observation_id":"aa78aa4f-a42b-476b-b5e6-c66dfc775739","resolution":{"observed_at":"2026-06-29T08:13:15.654177Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"cited_work":{"arxiv_id":"2509.02359","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2509.02359","snapshot_observed_at":"2026-07-04T19:50:10.271272Z","title":"Why do mllms struggle with spatial understanding? a systematic analysis from data to architecture","venue":null,"work_id":"0ff3d2c2-b3b3-41b0-8bb3-bae4d41de054","year":2025},"citing_paper":{"arxiv_id":"2606.11770","last_updated":"2026-06-10T07:54:42Z","snapshot_observed_at":"2026-07-06T23:50:49.040880Z","submitted_at":"2026-06-10T07:54:42Z","title":"SVoT: State-aware Visualization-of-Thought for Spatial Reasoning via Reinforcement Learning","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-27T09:59:02.899488Z"},"links":{"cited_paper":"/paper/2509.02359","citing_paper":"/paper/2606.11770"},"observation_digest":"sha256:d94fd3fa892abb1809a3af6932dcd5071621ff990af9b3d7bf9438c42555cc71","observation_id":"ed6f41d5-a745-45da-95ca-f5c128022165","resolution":{"observed_at":"2026-07-03T10:27:56.932385Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"cited_work":{"arxiv_id":"2509.02359","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2509.02359","snapshot_observed_at":"2026-07-04T19:50:10.271272Z","title":"Why do mllms struggle with spatial understanding? a systematic analysis from data to architecture","venue":null,"work_id":"0ff3d2c2-b3b3-41b0-8bb3-bae4d41de054","year":2025},"citing_paper":{"arxiv_id":"2606.25634","last_updated":"2026-06-24T09:38:27Z","snapshot_observed_at":"2026-07-07T00:00:02.505448Z","submitted_at":"2026-06-24T09:38:27Z","title":"SSMNBench: Diagnosing Image-based Cross-View Human-Object Understanding via Single-View Sufficiency and Multi-View Necessity","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-06-25T21:02:44.441202Z"},"links":{"cited_paper":"/paper/2509.02359","citing_paper":"/paper/2606.25634"},"observation_digest":"sha256:0c2a5761df41b0562c3da2ec3722db98a52fe3f97c7beb4f6066b91992fab737","observation_id":"c0e47212-e3d1-4316-8fe1-d51a03f0261d","resolution":{"observed_at":"2026-07-04T19:50:10.273001Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.02359","snapshot_observed_at":"2026-08-02T06:33:30.837244Z","title":"Why do mllms struggle with spatial understanding? a systematic analysis from data to architecture.arXiv preprint arXiv:2509.02359, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.12477","last_updated":"2026-07-15T01:51:44Z","snapshot_observed_at":"2026-08-13T20:13:00.240248Z","submitted_at":"2026-07-14T08:04:31Z","title":"Self in Space: Benchmarking Self-Awareness and Spatial Cognition in UAV Embodied Intelligence","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-02T06:33:30.837244Z"},"links":{"cited_paper":"/paper/2509.02359","citing_paper":"/paper/2607.12477"},"observation_digest":"sha256:6a050da1aea0006af20aa715e55d64f7921159ed70a701362de303588564dba1","observation_id":"800777b3-396f-4f56-848a-6d90e020a154","resolution":{"observed_at":"2026-08-02T06:33:30.837244Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2509.02359/citation-record","integrity":"/paper/2509.02359/integrity","json":"/paper/2509.02359/citation-record.json","paper":"/paper/2509.02359"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-05T11:40:20.917876Z","title":"L.; Almeida, D.; Altenschmidt, J.; Altman, S.; Anadkat, S.; et al","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:20.917876Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:e20cbc56f531ba51a4bba15258541da8420c9e1fa5c707b710b63336a674c103","observation_id":"d43cbc69-ffeb-4b7c-a279-03935eadac1f","resolution":{"observed_at":"2026-08-05T11:40:20.917876Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:26.180009Z","title":null,"venue":null,"work_id":"f09ef113-cc33-4ad1-945f-d9441bc9875d","year":2022},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:21.002919Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:f0f262a259aa42a4ed23ce287145e73725f9ae83338e8a105ffefe598eff0114","observation_id":"1fadd94f-a23d-4006-a4ca-231ea67b3297","resolution":{"observed_at":"2026-08-05T11:40:26.184222Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-08-14T04:17:22.593941Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.13923","snapshot_observed_at":"2026-08-05T11:40:21.062992Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:21.062992Z"},"links":{"cited_paper":"/paper/2502.13923","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:02d980852703750f8fe87977bac313c666060e5084487c5a8a4b2280ddb20ccd","observation_id":"c0d61a00-b585-478c-aabb-3d8897018d75","resolution":{"observed_at":"2026-08-05T11:40:21.062992Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:26.165175Z","title":"C.; Geva, M.; He, J.; Wu, J.; and Li, M","venue":null,"work_id":"04eb5202-ec0e-4b7f-b571-cc6cb3c7dbff","year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:21.124329Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:264ad65e2e8495dc7f9b4d042aa9d3343907d6c5b911997de913a07f03d2510a","observation_id":"9fdec59d-0450-4b24-9941-57f9a9bdc608","resolution":{"observed_at":"2026-08-05T11:40:26.169659Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.06192","last_updated":"2024-10-31T18:16:38Z","snapshot_observed_at":"2026-08-12T23:26:30.317326Z","submitted_at":"2024-07-08T17:59:57Z","title":"Multi-Object Hallucination in Vision-Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.06192","snapshot_observed_at":"2026-08-05T11:40:21.185699Z","title":"F.; and Chai, J","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:21.185699Z"},"links":{"cited_paper":"/paper/2407.06192","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:9cde05fa7e0b334247d3e96b379cd740c7a1131d6049ee1c3ae9810be3cc8fff","observation_id":"850362c7-dec3-4fde-8ee5-ebee38b93b0a","resolution":{"observed_at":"2026-08-05T11:40:21.185699Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:26.151489Z","title":null,"venue":null,"work_id":"bd659d37-cc25-4022-97a7-f19af24bb7e0","year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:21.262546Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:a0647ecea916bd8c8877c404117743539140c37f6ef2357934f2c9d34e9271d6","observation_id":"9f49fa7c-3e12-4290-aef7-767d57209960","resolution":{"observed_at":"2026-08-05T11:40:26.155733Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.12391","last_updated":"2025-07-16T16:37:13Z","snapshot_observed_at":"2026-08-12T02:28:25.913425Z","submitted_at":"2025-07-16T16:37:13Z","title":"Assessing the Value of Visual Input: A Benchmark of Multimodal Large Language Models for Robotic Path Planning","version":1},"cited_work":{"arxiv_id":"2507.12391","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.12391","snapshot_observed_at":"2026-08-05T11:40:25.831314Z","title":"Assessing the Value of Visual Input: A Benchmark of Multimodal Large Language Models for Robotic Path Planning","venue":"cs.RO","work_id":"e6e37e5d-76d1-4673-916d-5b43248e3707","year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:21.327031Z"},"links":{"cited_paper":"/paper/2507.12391","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:0b00ce6dd60929c66b237e88a62ffbbd1315fa3998f5a99c1d5fc93b66796b0b","observation_id":"ff00f254-9fce-4018-87cd-dad413f06d7e","resolution":{"observed_at":"2026-08-05T11:40:25.837858Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06500","last_updated":"2023-06-15T08:00:18Z","snapshot_observed_at":"2026-08-13T18:58:34.541884Z","submitted_at":"2023-05-11T00:38:10Z","title":"InstructBLIP: Towards General-purpose Vision-Language Models with Instruction Tuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06500","snapshot_observed_at":"2026-08-05T11:40:21.408465Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:21.408465Z"},"links":{"cited_paper":"/paper/2305.06500","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:2b5374281de4e3dfd5abca84bdb1c6352ff182c94d928f2cb968926f88789115","observation_id":"08d70087-a7c7-41a2-a144-805825c4e51f","resolution":{"observed_at":"2026-08-05T11:40:21.408465Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06788","last_updated":"2025-07-24T10:29:52Z","snapshot_observed_at":"2026-08-09T02:10:24.105469Z","submitted_at":"2025-02-10T18:59:58Z","title":"EVEv2: Improved Baselines for Encoder-Free Vision-Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.06788","snapshot_observed_at":"2026-08-05T11:40:21.474439Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:21.474439Z"},"links":{"cited_paper":"/paper/2502.06788","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:b995a7aeef5c71d1e2028a6889b6cb8938fe7aea203883a467b05b08ff16b286","observation_id":"3c7b937c-c260-4d00-b0f2-9190e4cf32e9","resolution":{"observed_at":"2026-08-05T11:40:21.474439Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-08-13T14:19:26.598265Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-05T11:40:21.539233Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:21.539233Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:6f5017d35c2fae9f5a403585d3fbb73a71e7ed6394e6da6d75cd05221333030a","observation_id":"a7546be4-08d4-4789-aea9-58029f5ab5c9","resolution":{"observed_at":"2026-08-05T11:40:21.539233Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:26.136030Z","title":null,"venue":null,"work_id":"ba6d09ed-0954-4ebe-b37d-c9feeb0ddbb3","year":2024},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:21.609690Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:fb565d84f575e685aa9732d2eae4c105965a087724aa92e5ceca3d8827ad6600","observation_id":"163291e9-bbce-4c0d-81d4-877fd88e0107","resolution":{"observed_at":"2026-08-05T11:40:26.141801Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:26.120105Z","title":null,"venue":null,"work_id":"7acc9299-7d5b-47f0-a62e-b19ae4c38644","year":2024},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:21.703116Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:9779da71ff7e557ed9ca450cbeb83684c70d594067bb323ae2f8c481fa85f019","observation_id":"c7ecdfc5-0bc4-4898-84be-61894d17628a","resolution":{"observed_at":"2026-08-05T11:40:26.124659Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.21776","last_updated":"2025-10-22T16:42:24Z","snapshot_observed_at":"2026-08-05T07:15:29.998948Z","submitted_at":"2025-03-27T17:59:51Z","title":"Video-R1: Reinforcing Video Reasoning in MLLMs","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.21776","snapshot_observed_at":"2026-08-05T11:40:21.757867Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:21.757867Z"},"links":{"cited_paper":"/paper/2503.21776","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:f0aa4f862e4eb8ad9e7eb5c05ed0d4d34cea8e578c67cdf4f004df36e12f3560","observation_id":"568b43a4-4766-4cfd-8a54-cf81b7105324","resolution":{"observed_at":"2026-08-05T11:40:21.757867Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:26.103909Z","title":null,"venue":null,"work_id":"d75dcb96-d8c9-498c-b84c-d0273cffe6ed","year":2024},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:21.847352Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:218a7edf6f85555fc1a4557e834ef99f1419bd2fbb0ef76748449d64cc58c92f","observation_id":"c3293966-faaa-4a51-aec1-94670e677bca","resolution":{"observed_at":"2026-08-05T11:40:26.108894Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:26.088365Z","title":"J.; Shen, Y.; Wallis, P.; Allen-Zhu, Z.; Li, Y.; Wang, S.; Wang, L.; Chen, W.; et al","venue":null,"work_id":"60e86211-0aa1-49e6-ba27-aa743b904426","year":2022},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:21.902977Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:682320879d0f6f6e81a3526cf1847e275cd923a5ee2d2daa03287318c49e3443","observation_id":"1db48a6c-7491-4470-a3f8-6bfa46112aa5","resolution":{"observed_at":"2026-08-05T11:40:26.093096Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-05T11:40:21.985438Z","title":"P.; Perelman, A.; Ramesh, A.; Clark, A.; Ostrow, A.; Welihinda, A.; Hayes, A.; Radford, A.; et al","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:21.985438Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:470aa9bb937c6d611bbf5afb7716d26ca40995dceebd0c8ee4972b396364d33e","observation_id":"f8d1a677-5f44-486a-b010-114fcca076dc","resolution":{"observed_at":"2026-08-05T11:40:21.985438Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:22.040115Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:22.040115Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:a88c3dc47d20c72f752ee0d2c3d7bc50a895c7ccc5968f1e42da33d084700988","observation_id":"fcc756ef-b175-4e56-92b9-42691d5bb884","resolution":{"observed_at":"2026-08-05T11:40:22.040115Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10167","last_updated":"2025-03-18T00:25:47Z","snapshot_observed_at":"2026-08-07T17:07:43.214031Z","submitted_at":"2025-03-13T08:46:32Z","title":"\"Well, Keep Thinking\": Enhancing LLM Reasoning with Adaptive Injection Decoding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10167","snapshot_observed_at":"2026-08-05T11:40:22.153841Z","title":"Well, Keep Thinking","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:22.153841Z"},"links":{"cited_paper":"/paper/2503.10167","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:c1e7b111884409d54fb7894728fa6c6fe2f4336588af247d3f2787b1bec3343f","observation_id":"0b96c9d4-86fa-4a0a-a84d-f15b19b42214","resolution":{"observed_at":"2026-08-05T11:40:22.153841Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:26.074297Z","title":null,"venue":null,"work_id":"61bad166-f7d9-4817-8d8a-f61fe5111a16","year":2023},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:22.201294Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:1dc2a06a540d4b585b493151d6501bfe80a42e2f7e17eb5586733ff211630855","observation_id":"9b2f27ef-2772-4c7b-9afd-8c80d6d9b29f","resolution":{"observed_at":"2026-08-05T11:40:26.078596Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.05474","last_updated":"2022-08-26T17:12:17Z","snapshot_observed_at":"2026-08-14T05:42:22.751765Z","submitted_at":"2017-12-14T23:17:24Z","title":"AI2-THOR: An Interactive 3D Environment for Visual AI","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.05474","snapshot_observed_at":"2026-08-05T11:40:22.265805Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:22.265805Z"},"links":{"cited_paper":"/paper/1712.05474","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:003e2f896707a119c6685bd3a05464c21aaffb99a756e5c67e3c7f55f6f406f8","observation_id":"321e73f9-a5cb-4c18-9f98-8102aeccbeeb","resolution":{"observed_at":"2026-08-05T11:40:22.265805Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:26.060269Z","title":"A.; et al","venue":null,"work_id":"1cc82f23-a2fe-4413-936d-077499470399","year":2017},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:22.336424Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:adde939b3b2f52d187c4289cff5102da807d7ac9b69eb4d58b5d246e6e258b7a","observation_id":"437e3c59-89c7-4fee-8faa-de9207f17962","resolution":{"observed_at":"2026-08-05T11:40:26.064579Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03326","last_updated":"2024-10-26T16:35:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:59:44Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03326","snapshot_observed_at":"2026-08-05T11:40:22.438908Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:22.438908Z"},"links":{"cited_paper":"/paper/2408.03326","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:4c69deae2cd9ea4e2cdf92f63a8410925aa37acbd905ad03cb14f2f7454f8cf7","observation_id":"c70964dc-271f-4a34-9fc1-3a19a1fa9d34","resolution":{"observed_at":"2026-08-05T11:40:22.438908Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:26.045528Z","title":null,"venue":null,"work_id":"7f633fcd-05c4-4387-adfc-c427c5e875ce","year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:22.541628Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:7c923819cf8a22d8d7ee25faf821dbfd69046383ef67632dab018f98fab1b9c4","observation_id":"e2e0e38c-2833-4be1-aa0f-1cecfe54f569","resolution":{"observed_at":"2026-08-05T11:40:26.049526Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.16217","last_updated":"2023-12-24T06:38:11Z","snapshot_observed_at":"2026-08-13T04:54:35.575920Z","submitted_at":"2023-12-24T06:38:11Z","title":"ManipLLM: Embodied Multimodal Large Language Model for Object-Centric Robotic Manipulation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.16217","snapshot_observed_at":"2026-08-05T11:40:22.609304Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:22.609304Z"},"links":{"cited_paper":"/paper/2312.16217","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:9c80fdfefc67b37ce14fe2f3aba03fc18cb39751cb4e8d89bf473333b58e77a8","observation_id":"6023ceb1-fe25-4dd7-9265-2df50acdd989","resolution":{"observed_at":"2026-08-05T11:40:22.609304Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.00883","last_updated":"2025-04-14T20:12:57Z","snapshot_observed_at":"2026-08-11T19:16:40.810461Z","submitted_at":"2025-04-01T15:11:11Z","title":"Improved Visual-Spatial Reasoning via R1-Zero-Like Training","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.00883","snapshot_observed_at":"2026-08-05T11:40:22.714354Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:22.714354Z"},"links":{"cited_paper":"/paper/2504.00883","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:235a00d51c04ceab3fbef54fbcda11fa0316f08938e66bc0ac165e576b85564c","observation_id":"c3a40141-58cc-4308-b02d-db22c41b6f97","resolution":{"observed_at":"2026-08-05T11:40:22.714354Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:22.773906Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:22.773906Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:b52e99c810d53030eb19e4d5de39fd16c87adad88fe07788ec532e1f5d3ba678","observation_id":"1e0205e3-8c82-4a78-993b-45b3ead780f1","resolution":{"observed_at":"2026-08-05T11:40:22.773906Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:22.835023Z","title":null,"venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:22.835023Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:dc4f617534149b7a68e696debb8a5e1eec61490db651dc656ad3d32190f0419a","observation_id":"ede1b8b4-47f2-4291-b58b-c48bb33fa726","resolution":{"observed_at":"2026-08-05T11:40:22.835023Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:22.893886Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:22.893886Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:2e5fb0f44f839fdedeaaa5aa73f788de1d8fb0d707167cf6d36332ed7c2e0491","observation_id":"19b493c7-82ac-4842-9a67-268051338242","resolution":{"observed_at":"2026-08-05T11:40:22.893886Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.08268","last_updated":"2025-02-03T21:47:31Z","snapshot_observed_at":"2026-08-14T09:16:47.522119Z","submitted_at":"2024-02-13T07:47:36Z","title":"World Model on Million-Length Video And Language With Blockwise RingAttention","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.08268","snapshot_observed_at":"2026-08-05T11:40:23.027800Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:23.027800Z"},"links":{"cited_paper":"/paper/2402.08268","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:3e111cbdaf878194745a5943fec04744b33b610f0d1ca07ede36045b9add2f95","observation_id":"6075b8c2-d58a-4fb9-975d-1df9b2df2341","resolution":{"observed_at":"2026-08-05T11:40:23.027800Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.10074","last_updated":"2025-01-23T02:31:25Z","snapshot_observed_at":"2026-08-10T21:32:39.307025Z","submitted_at":"2025-01-17T09:46:27Z","title":"SpatialCoT: Advancing Spatial Reasoning through Coordinate Alignment and Chain-of-Thought for Embodied Task Planning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.10074","snapshot_observed_at":"2026-08-05T11:40:23.101934Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:23.101934Z"},"links":{"cited_paper":"/paper/2501.10074","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:fba4cdb8352afacadc36758ae240851aab45bffa962b16301cf7169ceabe5f29","observation_id":"7ee2fe4f-6421-460a-b40b-9b26b89109b7","resolution":{"observed_at":"2026-08-05T11:40:23.101934Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:26.010703Z","title":null,"venue":null,"work_id":"36f4d3ee-c045-4ccf-a33a-0ae5aa8eb846","year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:23.186680Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:d0f7efc3eae9534295c7391d27c6909c3be54316bdcce66d0bb9e52fe0dfa94f","observation_id":"3f266c45-5168-4cfd-8d3d-37bd68848f6c","resolution":{"observed_at":"2026-08-05T11:40:26.014987Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.01805","last_updated":"2025-05-21T09:38:44Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-02T15:12:17Z","title":"SpaceR: Reinforcing MLLMs in Video Spatial Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.01805","snapshot_observed_at":"2026-08-05T11:40:23.287038Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:23.287038Z"},"links":{"cited_paper":"/paper/2504.01805","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:37679f1848f44c8319007e027d275b753bf918793b23dc9f67a1649091ca5d65","observation_id":"cbcd96db-b8ce-418d-b8bd-df236d371313","resolution":{"observed_at":"2026-08-05T11:40:23.287038Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2103.00020","last_updated":"2021-02-26T19:04:58Z","snapshot_observed_at":"2026-07-06T10:45:03.059688Z","submitted_at":"2021-02-26T19:04:58Z","title":"Learning Transferable Visual Models From Natural Language Supervision","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2103.00020","snapshot_observed_at":"2026-08-05T11:40:23.409237Z","title":"W.; Hallacy, C.; Ramesh, A.; Goh, G.; Agarwal, S.; Sastry, G.; Askell, A.; Mishkin, P.; Clark, J.; Krueger, G.; and Sutskever, I","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:23.409237Z"},"links":{"cited_paper":"/paper/2103.00020","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:7ca4ad41baa5897ff0a4f2f493e75148f6c68684e56d378ccb8f0b97fae32b8f","observation_id":"de4b4843-70f8-4554-b263-123776f5cb38","resolution":{"observed_at":"2026-08-05T11:40:23.409237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:25.996730Z","title":null,"venue":null,"work_id":"ebea23ff-d73e-4c28-82cf-0a8f4e482006","year":2024},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:23.532237Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:51bb5f762b6a45e6c3a8d5e3be8358bdb2e51ee110f898ae392762ab6c64e553","observation_id":"97553819-fd5a-44e4-9949-f61828487942","resolution":{"observed_at":"2026-08-05T11:40:26.000797Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:25.982741Z","title":null,"venue":null,"work_id":"673fdb87-b5ca-432e-b604-6e8cf78760ce","year":2024},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:23.645591Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:a151159b9a8cb8ad02846703328b18b8bb076f37947d3340cd9c1eee73dbfe75","observation_id":"bfdee345-9c44-4db4-8531-9b1bcd881ded","resolution":{"observed_at":"2026-08-05T11:40:25.987127Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:23.706928Z","title":"V.; Zhou, D.; et al","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:23.706928Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:6586158caf2f3aac62492882b1f1f4f3028538d22505e8290d8d7c8b543a8b35","observation_id":"f8855729-854d-4693-86b5-e93008d60952","resolution":{"observed_at":"2026-08-05T11:40:23.706928Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.05173","last_updated":"2025-05-30T03:54:16Z","snapshot_observed_at":"2026-08-13T06:46:27.328035Z","submitted_at":"2025-02-07T18:56:04Z","title":"VideoRoPE: What Makes for Good Video Rotary Position Embedding?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.05173","snapshot_observed_at":"2026-08-05T11:40:23.835237Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:23.835237Z"},"links":{"cited_paper":"/paper/2502.05173","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:0cc9471c290d218b3c323c3a2b7b64a09e781cc3b9c18068015c3fc4e20ed73b","observation_id":"403e88a9-0bf6-44aa-8100-3224b1e886ed","resolution":{"observed_at":"2026-08-05T11:40:23.835237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23747","last_updated":"2026-05-19T02:23:16Z","snapshot_observed_at":"2026-08-12T18:05:43.082727Z","submitted_at":"2025-05-29T17:59:04Z","title":"Spatial-MLLM: Boosting MLLM Capabilities in Visual-based Spatial Intelligence","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.23747","snapshot_observed_at":"2026-08-05T11:40:23.949207Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:23.949207Z"},"links":{"cited_paper":"/paper/2505.23747","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:0326bcc2475baa9cf8bc0d511c2a12abef2158aa94316575b08c2593630525a9","observation_id":"ef897e87-fbc2-49b5-94b3-bc2454b30aaa","resolution":{"observed_at":"2026-08-05T11:40:23.949207Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.12799","last_updated":"2025-03-24T11:30:58Z","snapshot_observed_at":"2026-08-12T01:00:02.203358Z","submitted_at":"2025-03-17T04:07:47Z","title":"Grounded Chain-of-Thought for Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.12799","snapshot_observed_at":"2026-08-05T11:40:24.078659Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:24.078659Z"},"links":{"cited_paper":"/paper/2503.12799","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:bdf75a9433649fd76c27ffe1ab10e537d8ff69dd5cf87d19d6d4f14aa146c586","observation_id":"f2774bf3-af99-407d-af29-21f65c5c471d","resolution":{"observed_at":"2026-08-05T11:40:24.078659Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:25.959089Z","title":null,"venue":null,"work_id":"2a9f1c29-2eec-44c5-8e9f-1e448aaaf780","year":2024},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:24.164678Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:da19da422cd2a621bb5d579dddf2bfa0ec77217dbe22767190e2c7a6cf088d69","observation_id":"d704a65f-af9c-4d3c-afde-1b14391faacc","resolution":{"observed_at":"2026-08-05T11:40:25.963261Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.12148","last_updated":"2023-12-19T13:31:24Z","snapshot_observed_at":"2026-08-13T04:58:35.298301Z","submitted_at":"2023-12-19T13:31:24Z","title":"Parameter-Efficient Fine-Tuning Methods for Pretrained Language Models: A Critical Review and Assessment","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.12148","snapshot_observed_at":"2026-08-05T11:40:24.298029Z","title":"J.; Tao, X.; and Wang, F","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:24.298029Z"},"links":{"cited_paper":"/paper/2312.12148","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:c304325a0de80c848b5a2eb54e7d02aa21762c8c33269a35a566919d2fda899d","observation_id":"b9820517-4e69-4ed2-8b76-c3d73ac9ccbe","resolution":{"observed_at":"2026-08-05T11:40:24.298029Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.14171","last_updated":"2025-07-02T21:00:36Z","snapshot_observed_at":"2026-08-10T08:02:13.965614Z","submitted_at":"2024-12-18T18:59:54Z","title":"Thinking in Space: How Multimodal Large Language Models See, Remember, and Recall Spaces","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.14171","snapshot_observed_at":"2026-08-05T11:40:24.437770Z","title":"W.; Han, R.; Fei-Fei, L.; and Xie, S","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:24.437770Z"},"links":{"cited_paper":"/paper/2412.14171","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:760e32e8c7c87a6fee9a422d3445714c6e3bda1a85c62c4171158f44707bf394","observation_id":"af90493a-917d-49dd-990a-54b38a3e4d24","resolution":{"observed_at":"2026-08-05T11:40:24.437770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:25.944909Z","title":null,"venue":null,"work_id":"f8e5a260-822f-41fe-983e-003f908d0042","year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:24.532215Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:99e69c75611300da172abb26a7a9bf1f7ecb62f60271e989fd06d389434b3180","observation_id":"5b34fe1e-369b-4e4f-8a31-1f4ff23b8b7d","resolution":{"observed_at":"2026-08-05T11:40:25.949071Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:25.930335Z","title":null,"venue":null,"work_id":"a25112a0-5aa0-410d-abb4-3ac576017e01","year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:24.634793Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:5b7c89fbc50d539f155385ffd4fcda4a91eca1a68aaa871b2f1b7a5f5026c14f","observation_id":"8956414d-be47-464a-8362-24872587b1fc","resolution":{"observed_at":"2026-08-05T11:40:25.934472Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.00493","last_updated":"2025-03-27T10:30:42Z","snapshot_observed_at":"2026-08-12T23:13:45.948951Z","submitted_at":"2024-11-30T14:28:53Z","title":"Video-3D LLM: Learning Position-Aware Video Representation for 3D Scene Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.00493","snapshot_observed_at":"2026-08-05T11:40:24.704510Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:24.704510Z"},"links":{"cited_paper":"/paper/2412.00493","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:26bceb8131c94474d0c532db0c35918e179127f4ce9306dca97c61cec6a5c48d","observation_id":"2d0cd3b5-f8a4-4fb0-95d8-c5b13fa3842e","resolution":{"observed_at":"2026-08-05T11:40:24.704510Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:25.915030Z","title":null,"venue":null,"work_id":"4016c94e-c7ab-49ba-89f3-20afd23b9f06","year":2024},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:24.841237Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:8dfe6ab2540515548302f24b0b6275cb2045c96f099eeca62e9c8aaf98e2b0bb","observation_id":"c9dd781a-e214-460d-9585-0f1e0eddc8b5","resolution":{"observed_at":"2026-08-05T11:40:25.919917Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.10479","last_updated":"2025-04-19T03:47:21Z","snapshot_observed_at":"2026-08-10T18:37:57.419939Z","submitted_at":"2025-04-14T17:59:25Z","title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.10479","snapshot_observed_at":"2026-08-05T11:40:24.932938Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:24.932938Z"},"links":{"cited_paper":"/paper/2504.10479","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:e3b5dec928de7f5c96d89d46f61884dd6e3e9ea23071ee5c44b5dd0cd80ad679","observation_id":"45e98c5a-18a5-468e-8f65-9ca4a5819e59","resolution":{"observed_at":"2026-08-05T11:40:24.932938Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:25.030350Z","title":", \" * write output.state after.block = add.period write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:25.030350Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:7116901d69e9cee08cab0a67a1e4ab0a08d0313fb6c5a9621eb05c24d4f7cb04","observation_id":"beec9b9e-010f-42fa-a22b-4ea66791e88f","resolution":{"observed_at":"2026-08-05T11:40:25.030350Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:25.147462Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:25.147462Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:7c934603ec2c5efccb5ce62122997738b1f096fcf387484689326e9a928a14fa","observation_id":"f1de502a-8bf3-41f3-8753-cc32b6943997","resolution":{"observed_at":"2026-08-05T11:40:25.147462Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-14T09:17:32.505825Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture"},"reference_resolution":{"displayed":49,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":45,"verified_exact":1,"verified_fuzzy":3},"total_outbound_references":49},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"thesis":"As of 14 August 2026, this Paper Citation Record lists 49 of 49 outbound references and 8 inbound Pith citation observations for arXiv:2509.02359."}