{"as_of":"2026-08-09T23:20:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b2b7ec85010dc3f7de86e1fb9e8e13e18fc7c130ae0bf553d0626dfadc0c42e1","coverage":[{"denominator":37,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":37,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:24:07.489391Z","state":"measured"},{"denominator":39,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":39,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-03T13:26:06.713886Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.19015","snapshot_observed_at":"2026-08-03T13:26:06.713886Z","title":"Can multi- modal large language models understand spatial relations? arXiv preprint arXiv:2505.19015, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2512.24331","last_updated":"2026-05-25T12:17:46Z","snapshot_observed_at":"2026-08-07T00:52:13.886460Z","submitted_at":"2025-12-30T16:35:00Z","title":"Spatial-aware Vision Language Model for Autonomous Driving","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-03T13:26:06.713886Z"},"links":{"cited_paper":"/paper/2505.19015","citing_paper":"/paper/2512.24331"},"observation_digest":"sha256:9399cabd2184b79c75eaec3343f301d1fd0776922aa0693678309d55abb657af","observation_id":"09564187-549e-4c2d-84e4-0eef8ac830b1","resolution":{"observed_at":"2026-08-03T13:26:06.713886Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.19015","snapshot_observed_at":"2026-08-03T12:25:50.226593Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2601.03191","last_updated":"2026-05-23T10:05:21Z","snapshot_observed_at":"2026-08-09T16:56:17.744290Z","submitted_at":"2026-01-06T17:13:23Z","title":"AnatomiX, an Anatomy-Aware Grounded Multimodal Large Language Model for Chest X-Ray Interpretation","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-03T12:25:50.226593Z"},"links":{"cited_paper":"/paper/2505.19015","citing_paper":"/paper/2601.03191"},"observation_digest":"sha256:c8ee5126320872cb430484c2200f62f3b4cee5680377140d72f2f41027553e97","observation_id":"e795f2a0-0f3f-4d36-9c85-8229c6859310","resolution":{"observed_at":"2026-08-03T12:25:50.226593Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2505.19015/citation-record","integrity":"/paper/2505.19015/integrity","json":"/paper/2505.19015/citation-record.json","paper":"/paper/2505.19015"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:04.016067Z","title":"online\" 'onlinestring :=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:04.016067Z"},"links":{"citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:e9778528ad9016c90251a486c654b0f1803bf26ab348e692175fb98814f4b548","observation_id":"94638c2e-606d-4dbb-9e17-61e2c405cf06","resolution":{"observed_at":"2026-08-07T14:24:04.016067Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:04.199896Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:04.199896Z"},"links":{"citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:d0db7faa636c72b74301bf82db9a24de023a348734521613bdb22842f0ceefe4","observation_id":"9ea1da04-f617-4fe9-8880-a1f75defe4b7","resolution":{"observed_at":"2026-08-07T14:24:04.199896Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-07T14:24:04.305365Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:04.305365Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:a1f2645b24ff9b966a27b16862614643a97607bbfc91b15d62a901c81048f5b0","observation_id":"85c76fd1-5255-4319-9c6b-f442c04bdd16","resolution":{"observed_at":"2026-08-07T14:24:04.305365Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:10.119663Z","title":null,"venue":null,"work_id":"45abd6b1-e672-43c8-b698-b1c5460f18b9","year":2018},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:04.434248Z"},"links":{"citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:459c4a651b666b9e4f98f53ce28703483898a87b781f11df8c206986f87667ea","observation_id":"04321268-c26f-4c1a-8693-d555a8a2fd45","resolution":{"observed_at":"2026-08-07T14:24:10.227755Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-09T16:57:03.077389Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-07T14:24:04.556539Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:04.556539Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:b03b8ddcb6cd309f7bb3fdebe13573840e45c45ac708aa6de0565bf9ea9b4056","observation_id":"7b406e20-d445-4681-a415-e8c4673aa066","resolution":{"observed_at":"2026-08-07T14:24:04.556539Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:04.683366Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:04.683366Z"},"links":{"citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:9a1021a05a96a15bad401cf5195ba1573bfc1da2196b6eeca9cc874863801d66","observation_id":"71a025e6-50ec-4938-aa1a-ffaeb4266c67","resolution":{"observed_at":"2026-08-07T14:24:04.683366Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01584","last_updated":"2024-10-15T01:16:20Z","snapshot_observed_at":"2026-08-09T16:58:20.916397Z","submitted_at":"2024-06-03T17:59:06Z","title":"SpatialRGPT: Grounded Spatial Reasoning in Vision Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01584","snapshot_observed_at":"2026-08-07T14:24:04.777438Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:04.777438Z"},"links":{"cited_paper":"/paper/2406.01584","citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:1ed5754615c4ef4d2a6c9add1bfe8c0d7f26673c3284d879f321ee560e6c5c86","observation_id":"6ee4ee5c-9a39-4345-90d7-00518300fbf6","resolution":{"observed_at":"2026-08-07T14:24:04.777438Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.15596","last_updated":"2024-03-28T11:35:55Z","snapshot_observed_at":"2026-08-09T16:57:24.482436Z","submitted_at":"2023-11-27T07:44:25Z","title":"EgoThink: Evaluating First-Person Perspective Thinking Capability of Vision-Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.15596","snapshot_observed_at":"2026-08-07T14:24:04.861523Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:04.861523Z"},"links":{"cited_paper":"/paper/2311.15596","citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:7b87623c2f45b6470c7d44384641acea408be68a5b2e8dc6f97b0560b9d3a065","observation_id":"5910c5c7-c0b6-417f-b3f8-26090edfbe5c","resolution":{"observed_at":"2026-08-07T14:24:04.861523Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:04.990570Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:04.990570Z"},"links":{"citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:5907cd6abe4d1e4d07952537e173dd6474cb0c3e596908850311339858ccd94d","observation_id":"e5b6b00a-112f-4fb0-b7bb-d1d5ba9207a0","resolution":{"observed_at":"2026-08-07T14:24:04.990570Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05756","last_updated":"2024-06-09T12:23:14Z","snapshot_observed_at":"2026-08-09T16:58:30.430469Z","submitted_at":"2024-06-09T12:23:14Z","title":"EmbSpatial-Bench: Benchmarking Spatial Understanding for Embodied Tasks with Large Vision-Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05756","snapshot_observed_at":"2026-08-07T14:24:05.093854Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:05.093854Z"},"links":{"cited_paper":"/paper/2406.05756","citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:a8fd01296aed60cb1bab4568458affdabec566358297be48b87dac5f7927a55b","observation_id":"208b9e5c-4569-46e9-bf2a-e06398c05537","resolution":{"observed_at":"2026-08-07T14:24:05.093854Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.13394","last_updated":"2025-10-24T02:45:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-23T09:22:36Z","title":"MME: A Comprehensive Evaluation Benchmark for Multimodal Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.13394","snapshot_observed_at":"2026-08-07T14:24:05.175871Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:05.175871Z"},"links":{"cited_paper":"/paper/2306.13394","citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:8c93794485ee47ff8b219c2b88c8efe9d6b49ebd60d4f369b8c071077bc7ef62","observation_id":"24c57652-fcc8-46e0-ae5e-2bb6da3ed56b","resolution":{"observed_at":"2026-08-07T14:24:05.175871Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01105","last_updated":"2024-09-05T03:38:08Z","snapshot_observed_at":"2026-07-06T17:24:03.237617Z","submitted_at":"2024-02-02T02:44:59Z","title":"A Survey for Foundation Models in Autonomous Driving","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01105","snapshot_observed_at":"2026-08-07T14:24:05.264871Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:05.264871Z"},"links":{"cited_paper":"/paper/2402.01105","citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:f917f58a6eb4775d53a3852b7eee0c124b9d009e3bc2d86c85bf9cece283b5fa","observation_id":"cd04fc05-1c0b-432d-b845-c8eabd87d837","resolution":{"observed_at":"2026-08-07T14:24:05.264871Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:09.878346Z","title":null,"venue":null,"work_id":"cabd8657-7386-4608-8bce-5d1a633e9a9c","year":2020},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:05.387147Z"},"links":{"citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:49c4fbb9db5289435fde45161035518008bfa384a0c79972c75f7038036248e3","observation_id":"652cca04-34ff-4909-9b8d-335a74b4ecfa","resolution":{"observed_at":"2026-08-07T14:24:09.988251Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:09.695051Z","title":null,"venue":null,"work_id":"72f8a8fc-f32b-4912-98c6-4ab869016ab0","year":2023},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:05.520208Z"},"links":{"citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:c32167fa3ec6b6bc7358dea7f2397537c995d5d5e9bf199b7f257f1ee89c3c2c","observation_id":"cbed56f3-c34d-4f67-965c-dc247beb41c2","resolution":{"observed_at":"2026-08-07T14:24:09.741763Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:09.522539Z","title":null,"venue":null,"work_id":"33365e32-145a-47e6-84bc-a5e076bcfadd","year":2020},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:05.606345Z"},"links":{"citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:0ef689fe7c2fb2a3d47ee56bd286ddbc6b2bf68271b66cb648bea543321ab458","observation_id":"fec98b1a-2dcc-4e76-8fe1-e6d477ccd908","resolution":{"observed_at":"2026-08-07T14:24:09.579034Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.09685","last_updated":"2021-10-16T18:40:34Z","snapshot_observed_at":"2026-08-07T07:43:16.294957Z","submitted_at":"2021-06-17T17:37:18Z","title":"LoRA: Low-Rank Adaptation of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.09685","snapshot_observed_at":"2026-08-07T14:24:05.718317Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:05.718317Z"},"links":{"cited_paper":"/paper/2106.09685","citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:f575ff7a99ffff3339079318bdb44e68d423c6a27cee2d6ee0f9e10f41aa72fb","observation_id":"1773f94f-d438-4a2a-bb74-d5b2755fb262","resolution":{"observed_at":"2026-08-07T14:24:05.718317Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:05.842198Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:05.842198Z"},"links":{"citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:98f29e5d9575746c563a988112b2077718fa136f7fdbfbedda0ccb7510afb88c","observation_id":"a0d41e7b-d23b-4c9d-bd9c-4de51431891a","resolution":{"observed_at":"2026-08-07T14:24:05.842198Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:09.323063Z","title":null,"venue":null,"work_id":"49a4ecab-f383-4b1d-b95d-ee3f25584240","year":2017},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:05.957479Z"},"links":{"citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:9dc3a31ac93a3af7c0ae4a48ccc701ae01f2b400d433ad38540f93af5168b6bc","observation_id":"9eace458-3a77-43b0-9f48-a1c7af94ac54","resolution":{"observed_at":"2026-08-07T14:24:09.427448Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:09.155026Z","title":null,"venue":null,"work_id":"9dc30b15-edb9-4a94-88a0-bfeb131b377e","year":2023},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:06.041764Z"},"links":{"citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:75096298e150b7c6816403a8cd55a0d8bbb4a2a942b7609540e491ad63280893","observation_id":"55cdf097-1c50-46a1-bfae-b9cfa6a8fcd9","resolution":{"observed_at":"2026-08-07T14:24:09.241890Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:06.159421Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:06.159421Z"},"links":{"citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:2d5abca8c392d099368b6864722fdf4ec01d6e07e63fb281e22b6cae9282e6f3","observation_id":"ed5c6e44-bc67-476c-8aab-6c311db9b9c8","resolution":{"observed_at":"2026-08-07T14:24:06.159421Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:06.250116Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:06.250116Z"},"links":{"citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:4391b274dcf26d0da109bd68a45b5ab2fc5ffadb3b1b2eee07075a9eefb4563d","observation_id":"412c00e5-dc90-46ec-b90f-1bc987e3c628","resolution":{"observed_at":"2026-08-07T14:24:06.250116Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:06.318633Z","title":null,"venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:06.318633Z"},"links":{"citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:ffc7fe0c2c3cdea3525b886f5c71ec56a27512e0e55b55673fc078f2e3b950ff","observation_id":"159aa6df-2de7-4683-a052-955e6dcee207","resolution":{"observed_at":"2026-08-07T14:24:06.318633Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:08.927024Z","title":null,"venue":null,"work_id":"1827e7b7-59a5-4f70-922a-a89d945833a7","year":2023},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:06.384825Z"},"links":{"citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:9fcee58e851943cf805735ba9153e993928ee92ef0c192472ecb572b1a6bb3e3","observation_id":"93aa9eee-a76a-4107-9616-0622e863650c","resolution":{"observed_at":"2026-08-07T14:24:09.026833Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:06.425516Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:06.425516Z"},"links":{"citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:272e86cd43027ad8122345e844e3f29076c51041b933dda0f9a3c99bd68a6da0","observation_id":"0b6a188c-c912-4935-9f2f-77a8edfa9679","resolution":{"observed_at":"2026-08-07T14:24:06.425516Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.04790","last_updated":"2023-09-09T13:35:01Z","snapshot_observed_at":"2026-08-09T16:57:23.628046Z","submitted_at":"2023-09-09T13:35:01Z","title":"MMHQA-ICL: Multimodal In-context Learning for Hybrid Question Answering over Text, Tables and Images","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.04790","snapshot_observed_at":"2026-08-07T14:24:06.523784Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:06.523784Z"},"links":{"cited_paper":"/paper/2309.04790","citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:8fb04ef0b4b3f12b9fb5bd9e8252b4faa3ee98d2851373e5fe7e96bc6f1e94ed","observation_id":"1d0ad457-f85f-435c-a485-8344616daf1e","resolution":{"observed_at":"2026-08-07T14:24:06.523784Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.07602","last_updated":"2022-03-20T15:13:08Z","snapshot_observed_at":"2026-08-08T11:15:37.346201Z","submitted_at":"2021-10-14T17:58:47Z","title":"P-Tuning v2: Prompt Tuning Can Be Comparable to Fine-tuning Universally Across Scales and Tasks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.07602","snapshot_observed_at":"2026-08-07T14:24:06.627542Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:06.627542Z"},"links":{"cited_paper":"/paper/2110.07602","citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:f1cb721081024d75c604f188f24ea30a869527cdce0ac4b967bd21fbd82ad977","observation_id":"8fbdc9b1-e232-4dd3-807e-8d2afa261895","resolution":{"observed_at":"2026-08-07T14:24:06.627542Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:08.732282Z","title":null,"venue":null,"work_id":"9f43ac3e-789c-4d31-958d-7d2948798ebc","year":2016},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:06.716472Z"},"links":{"citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:ddddf47aacf86605d1589df5d3258668a203d87444f8437f10a1cf09268493cb","observation_id":"7f272d00-8eeb-4138-8f64-c735d3004cb6","resolution":{"observed_at":"2026-08-07T14:24:08.828915Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:08.580875Z","title":null,"venue":null,"work_id":"281ea923-92d2-4d6f-87cc-3f6f63cd90c7","year":2023},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:06.800214Z"},"links":{"citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:326ef57357e6ac2d78899870da7948a74dcc380835447f74cc51f7a6fe7bba34","observation_id":"37e2bfd1-b1d1-4f4b-be8b-4daff3631f8c","resolution":{"observed_at":"2026-08-07T14:24:08.657963Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:08.439080Z","title":null,"venue":null,"work_id":"f803d67b-1452-4fec-afac-846bf982c5f3","year":2017},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:06.908390Z"},"links":{"citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:f0f4fc7cde646d31d8423c3f7fa970eccb6015aedb6130d88cbc04653253cd41","observation_id":"ca18be29-2106-47db-b102-f380972139fe","resolution":{"observed_at":"2026-08-07T14:24:08.500757Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-07T14:24:06.999213Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:06.999213Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:5bf10245c6c5cf849c148bd0c911985386349c5d28e02438591b23d56d3e4961","observation_id":"15815e6a-a2f6-437f-b5aa-4257a1c7f842","resolution":{"observed_at":"2026-08-07T14:24:06.999213Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:08.317358Z","title":null,"venue":null,"work_id":"725df56e-1f95-4633-a93c-438a48ef5e3d","year":2023},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:07.055340Z"},"links":{"citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:d1fe3cfacb1dca608f83a3f6354d21f836e0f2cdf7665946ff919c29b20ad7a5","observation_id":"1bfa95ca-ef8e-4085-91ad-ba81db48bbc0","resolution":{"observed_at":"2026-08-07T14:24:08.369030Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:08.161688Z","title":null,"venue":null,"work_id":"0cda1959-7558-4aec-a0ff-417d4837f6ac","year":2024},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:07.117272Z"},"links":{"citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:6dda891058ee946e90c17f4768ece6533b56ecd931d89ababc3d7c4d9cc28929","observation_id":"e4ad624f-e49c-4527-8863-40e463aac19b","resolution":{"observed_at":"2026-08-07T14:24:08.228301Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00729","last_updated":"2024-03-01T18:25:26Z","snapshot_observed_at":"2026-08-09T16:58:29.893740Z","submitted_at":"2024-03-01T18:25:26Z","title":"Can Transformers Capture Spatial Relations between Objects?","version":1},"cited_work":{"arxiv_id":"2403.00729","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.00729","snapshot_observed_at":"2026-08-07T14:24:07.658151Z","title":"Can Transformers Capture Spatial Relations between Objects?","venue":"cs.CV","work_id":"8bb069e3-6d7f-4518-8242-ad86d5ca7845","year":2024},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:07.210177Z"},"links":{"cited_paper":"/paper/2403.00729","citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:c267287617fde1932ed560d841985b71c9bebda0c49189c0b520d2856222f615","observation_id":"6d4bca06-c6b1-4636-aa82-31f910ef2921","resolution":{"observed_at":"2026-08-07T14:24:07.735657Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:08.001275Z","title":null,"venue":null,"work_id":"6fe2a8b0-eae5-4a46-84bd-752d29f15c7f","year":2019},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:07.281480Z"},"links":{"citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:4dd18f71f5553801711d922273699cfe7a3c09535a764861fcac02c995c84452","observation_id":"31e9da66-8395-45cb-ace7-d3729ca6b859","resolution":{"observed_at":"2026-08-07T14:24:08.054120Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14178","last_updated":"2024-03-29T08:13:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-27T13:27:01Z","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14178","snapshot_observed_at":"2026-08-07T14:24:07.339557Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:07.339557Z"},"links":{"cited_paper":"/paper/2304.14178","citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:5252bb3074ee579a1bda3f7a25c6ae147830a7caa4e287db3f0b732375ec25a3","observation_id":"9bef279a-abe8-4ce1-a3fc-342b17aa70ff","resolution":{"observed_at":"2026-08-07T14:24:07.339557Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.02582","last_updated":"2024-01-05T00:26:07Z","snapshot_observed_at":"2026-08-09T16:57:52.708316Z","submitted_at":"2024-01-05T00:26:07Z","title":"CoCoT: Contrastive Chain-of-Thought Prompting for Large Multimodal Models with Multiple Image Inputs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.02582","snapshot_observed_at":"2026-08-07T14:24:07.427455Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:07.427455Z"},"links":{"cited_paper":"/paper/2401.02582","citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:161351cdf2c3c1701b0dce34f2318661f6fe82f2f3a6c5947fff5e790cce70ac","observation_id":"8670db17-2935-4c00-b800-55c2984d1be3","resolution":{"observed_at":"2026-08-07T14:24:07.427455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:24:07.489391Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?","version":2},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-07T14:24:07.489391Z"},"links":{"citing_paper":"/paper/2505.19015"},"observation_digest":"sha256:bd04b9925ee0d492c619a54f6bd78941e8e4ed69775e27b470ca9bb123b20fa7","observation_id":"52c7551b-2261-42e7-b9d5-ad0aadc87c5d","resolution":{"observed_at":"2026-08-07T14:24:07.489391Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2505.19015","last_updated":"2025-08-08T09:55:18Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-09T16:55:25.490719Z","submitted_at":"2025-05-25T07:37:34Z","title":"Can Multimodal Large Language Models Understand Spatial Relations?"},"reference_resolution":{"displayed":37,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":36,"verified_exact":1,"verified_fuzzy":0},"total_outbound_references":37},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 37 of 37 outbound references and 2 inbound Pith citation observations for arXiv:2505.19015."}