{"as_of":"2026-08-21T07:51:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a4a599df50890e0978a73ce8796ad22267501b7f1772fb66a2c8e6dee3f75bc9","coverage":[{"denominator":49,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":49,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T05:29:57.069202Z","state":"measured"},{"denominator":52,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":52,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-21T06:32:19.484+00:00","state":"measured"},{"denominator":3,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":3,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-03T23:08:52.541313Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T08:47:48.085852Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20648","snapshot_observed_at":"2026-08-03T23:08:52.541313Z","title":"Spare: Enhancing spatial reasoning in vision-language models with synthetic data.arXiv preprint arXiv:2504.20648,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2511.07403","last_updated":"2026-07-02T19:21:24Z","snapshot_observed_at":"2026-08-20T12:06:49.196876Z","submitted_at":"2025-11-10T18:52:47Z","title":"SpatialThinker: Reinforcing Scene Graph-Grounded Spatial Reasoning via Dense Rewards","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-03T23:08:52.541313Z"},"links":{"cited_paper":"/paper/2504.20648","citing_paper":"/paper/2511.07403"},"observation_digest":"sha256:4ae8d27b259121c74208ece0efe81b03d358843d1522ac595132dcc71d86d387","observation_id":"a06ed670-65f6-48a4-991d-add349d902c4","resolution":{"observed_at":"2026-08-03T23:08:52.541313Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20648","snapshot_observed_at":"2026-08-03T16:55:21.645845Z","title":"Spare: Enhancing spatial rea- soning in vision-language models with synthetic data.arXiv preprint arXiv:2504.20648, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2512.11393","last_updated":"2026-08-13T11:40:44Z","snapshot_observed_at":"2026-08-16T23:11:50.762075Z","submitted_at":"2025-12-12T09:07:21Z","title":"The N-Body Problem: Parallel Execution from Single-Person Egocentric Video","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-03T16:55:21.645845Z"},"links":{"cited_paper":"/paper/2504.20648","citing_paper":"/paper/2512.11393"},"observation_digest":"sha256:17cd5bfa26156e03539cea2bc44103469beb31bde4a8deac1a700bc1b777a074","observation_id":"140de5f2-2d37-4963-b297-1a25da9ad0d5","resolution":{"observed_at":"2026-08-03T16:55:21.645845Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"cited_work":{"arxiv_id":"2504.20648","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.20648","snapshot_observed_at":"2026-07-03T08:47:48.085852Z","title":"Spare: Enhancing spa- tial reasoning in vision-language models with syn- thetic data.arXiv preprint arXiv:2504.20648, 2025","venue":null,"work_id":"a5434a8e-6cd5-485c-98f4-0de091328061","year":2025},"citing_paper":{"arxiv_id":"2606.11683","last_updated":"2026-06-10T05:52:14Z","snapshot_observed_at":"2026-08-17T03:39:46.869507Z","submitted_at":"2026-06-10T05:52:14Z","title":"Reason, Then Re-reason: Cross-view Revisiting Improves Spatial Reasoning","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-27T10:41:41.697216Z"},"links":{"cited_paper":"/paper/2504.20648","citing_paper":"/paper/2606.11683"},"observation_digest":"sha256:6578bf2a11152ccfd16f3d34cbb70b5461d1e6220f6b03a061f9bb8047e46235","observation_id":"b50d233f-b653-4633-a0c0-57dd2445ed4c","resolution":{"observed_at":"2026-07-03T08:47:48.087301Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2504.20648/citation-record","integrity":"/paper/2504.20648/integrity","json":"/paper/2504.20648/citation-record.json","paper":"/paper/2504.20648"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2404.14219","last_updated":"2024-08-30T21:17:17Z","snapshot_observed_at":"2026-08-17T03:25:04.404839Z","submitted_at":"2024-04-22T14:32:33Z","title":"Phi-3 Technical Report: A Highly Capable Language Model Locally on Your Phone","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.14219","snapshot_observed_at":"2026-08-16T05:29:56.758684Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.758684Z"},"links":{"cited_paper":"/paper/2404.14219","citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:4e808e974259f7fff77330fb487932bb235145a992ebba195094b946c41a1d37","observation_id":"4d287ad1-b9ba-4b86-9867-e05272372df8","resolution":{"observed_at":"2026-08-16T05:29:56.758684Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-17T09:58:46.058102Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-16T05:29:56.765764Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.765764Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:a56586e5b5d8909c481388cec6aec44b27545af9422a2e45f20af551e37c029b","observation_id":"de6a5bbb-00e5-4252-b77a-d93fdccc964c","resolution":{"observed_at":"2026-08-16T05:29:56.765764Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06680","last_updated":"2025-02-27T13:32:44Z","snapshot_observed_at":"2026-08-16T15:01:10.681714Z","submitted_at":"2023-09-13T02:35:59Z","title":"STUPD: A Synthetic Dataset for Spatial and Temporal Relation Reasoning","version":3},"cited_work":{"arxiv_id":"2309.06680","doi":null,"metadata_source":"pith","pith_arxiv_id":"2309.06680","snapshot_observed_at":"2026-08-16T05:29:57.697944Z","title":"STUPD: A Synthetic Dataset for Spatial and Temporal Relation Reasoning","venue":"cs.CV","work_id":"8328d372-b539-4458-9236-852f991197ff","year":2023},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.772171Z"},"links":{"cited_paper":"/paper/2309.06680","citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:70d830e59b760433f754a0bd91e97603eff4f7a48a6e697944375d1de6450208","observation_id":"29c621af-879a-43f8-ab05-4d5e8e84ac14","resolution":{"observed_at":"2026-08-16T05:29:57.707017Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:58.646220Z","title":"Balakrishnan, M.Syed Shahul Hameed, Kavya Venkatesan, and G Aswin","venue":null,"work_id":"71d31039-c593-4bb3-a2c4-603ec1e522ab","year":2021},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.778987Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:ba9977b6b83467e09db674d9fe0b56f0390ae696511c875f5543953b9150a779","observation_id":"a4047b22-be75-48c4-9dbb-b861740d1110","resolution":{"observed_at":"2026-08-16T05:29:58.652580Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:58.620252Z","title":null,"venue":null,"work_id":"5ab51f25-ed61-45c3-94c1-6748812cb43c","year":2023},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.784460Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:62eff96ddacb649ef63e7336103f1c3e124713abae55543a192cf0a4ffbe7fd7","observation_id":"b45aff58-74b9-44ab-b478-19005ff864b8","resolution":{"observed_at":"2026-08-16T05:29:58.626475Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:56.790350Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.790350Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:fff2be74f56bd44feabb3199bb831b109eb415d52776376900b03e4f46f1cf56","observation_id":"3cffdbe0-0758-4959-824a-b50ca4b231ef","resolution":{"observed_at":"2026-08-16T05:29:56.790350Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:56.798981Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.798981Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:bdcc98472e9ceaff38199e1b721cd581bfc981ac777052c269539beea30641e5","observation_id":"ec9b47b8-4e46-4bdf-a09b-5b012a4ac22b","resolution":{"observed_at":"2026-08-16T05:29:56.798981Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:56.811246Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.811246Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:036ab68a4df8b6da6d42b694e31d81ceba24b1ebc629fe1c3d8fc2d697517714","observation_id":"f341fa49-f9f2-4624-ad6d-94f67c6ef332","resolution":{"observed_at":"2026-08-16T05:29:56.811246Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:58.531246Z","title":null,"venue":null,"work_id":"55d527b3-fff3-4fee-90f3-0bcd968541b0","year":2022},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.817920Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:fbb8025909249bb7c582b8a42f95d126c4ec024f2f0229a14154175491c62a6b","observation_id":"c3684b0f-c005-409a-b015-9d7cbe067b16","resolution":{"observed_at":"2026-08-16T05:29:58.540028Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.17146","last_updated":"2024-12-05T14:28:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-25T17:59:51Z","title":"Molmo and PixMo: Open Weights and Open Data for State-of-the-Art Vision-Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.17146","snapshot_observed_at":"2026-08-16T05:29:56.824252Z","title":"Smith, Hannaneh Hajishirzi, Ross Girshick, Ali Farhadi, and Aniruddha Kembhavi","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.824252Z"},"links":{"cited_paper":"/paper/2409.17146","citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:28bc144d893d38934d542946393189be6789f320416f02b1a8c99d36643ad5b8","observation_id":"92011c0b-4a40-4a20-8489-9a681a8eb741","resolution":{"observed_at":"2026-08-16T05:29:56.824252Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:56.830331Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.830331Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:b530e8ff2a76a2146dcb62e39ac2301ed57fdd94577462bc6e3c4ac884c54cf9","observation_id":"9876694b-2bb8-4211-bf27-d15df6ae7976","resolution":{"observed_at":"2026-08-16T05:29:56.830331Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.10015","last_updated":"2023-10-27T17:24:04Z","snapshot_observed_at":"2026-08-16T16:07:26.772862Z","submitted_at":"2022-12-20T06:03:51Z","title":"Benchmarking Spatial Relationships in Text-to-Image Generation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.10015","snapshot_observed_at":"2026-08-16T05:29:56.836265Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.836265Z"},"links":{"cited_paper":"/paper/2212.10015","citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:cda3c565e4dbdabbc540e82f4e58b0c6eb2ca7034bfed29a9fae004d3f5d898f","observation_id":"5e76737f-d4ed-4270-89b3-0ca6aae8337f","resolution":{"observed_at":"2026-08-16T05:29:56.836265Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:56.842256Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.842256Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:7ea60b2886b03b5ac1b142288bdc5bd7215fd52022193ce8677c41ed4bd7dbad","observation_id":"58b8aa4d-fee0-41d8-967d-8b26263d14be","resolution":{"observed_at":"2026-08-16T05:29:56.842256Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:58.447726Z","title":null,"venue":null,"work_id":"1762f843-3b38-4e02-ad5f-903a223bb351","year":2024},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.848479Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:4063bde67e5dd9b6cce93a1892016ec63627db750694661b15f71216e8069e77","observation_id":"8eea6dc1-403a-4041-b94f-488b0292e503","resolution":{"observed_at":"2026-08-16T05:29:58.462650Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.11644","last_updated":"2023-10-02T06:12:30Z","snapshot_observed_at":"2026-08-13T11:19:55.436754Z","submitted_at":"2023-06-20T16:14:25Z","title":"Textbooks Are All You Need","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.11644","snapshot_observed_at":"2026-08-16T05:29:56.854784Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.854784Z"},"links":{"cited_paper":"/paper/2306.11644","citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:00b4af344c4f95de1b119779b0f8cfcec2dc752134d5b14878e1b36baa6d2c11","observation_id":"1648cafa-c215-4e2e-b39a-b384c934c0c5","resolution":{"observed_at":"2026-08-16T05:29:56.854784Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.08718","last_updated":"2022-03-23T19:47:21Z","snapshot_observed_at":"2026-07-06T11:01:02.207193Z","submitted_at":"2021-04-18T05:00:29Z","title":"CLIPScore: A Reference-free Evaluation Metric for Image Captioning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.08718","snapshot_observed_at":"2026-08-16T05:29:56.860744Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.860744Z"},"links":{"cited_paper":"/paper/2104.08718","citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:f60e9b296b265f08a7a680c3ca3bb7ae5c4486ec4157c7de70ab0c6b2bfa7f96","observation_id":"86914848-7193-435e-b39c-e8402835e93c","resolution":{"observed_at":"2026-08-16T05:29:56.860744Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.09685","last_updated":"2021-10-16T18:40:34Z","snapshot_observed_at":"2026-08-20T11:47:17.477107Z","submitted_at":"2021-06-17T17:37:18Z","title":"LoRA: Low-Rank Adaptation of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.09685","snapshot_observed_at":"2026-08-16T05:29:56.866451Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.866451Z"},"links":{"cited_paper":"/paper/2106.09685","citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:a01e3d70dca6447724f66435a0afa4d42af8fbb19b72db55fd2c89bf12fdb96e","observation_id":"110c6dd2-bc99-4bfa-9682-d6ae616aedd1","resolution":{"observed_at":"2026-08-16T05:29:56.866451Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:56.873302Z","title":"Hudson and Christopher D","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.873302Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:3c845ff61642b537fbf2b97116a2e5c5fefbd34417d1ed4e772982b56a1df25b","observation_id":"2c3c1928-14ef-437c-91ed-59c28c828366","resolution":{"observed_at":"2026-08-16T05:29:56.873302Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:56.879400Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.879400Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:b6a0073f65537b5a738a5a665f803cafbe60e3c0f4c8035b3eb5fea45cd3beab","observation_id":"2fdc0d94-ab46-4add-882f-3485cfc117ae","resolution":{"observed_at":"2026-08-16T05:29:56.879400Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:56.887748Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.887748Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:d35634c740b5eab47bb5425588b921bf28dec749876cbfdce41e0461129b129d","observation_id":"8149a53c-3e4f-4a99-b339-6ec8f2f5949c","resolution":{"observed_at":"2026-08-16T05:29:56.887748Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:56.893484Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.893484Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:4157a5cf6ef3d9b1d83bff490b3232bdf8a47d82a14af35998bf977f29346ab3","observation_id":"d49d6acb-d7e5-4637-a68e-1297ce093fb7","resolution":{"observed_at":"2026-08-16T05:29:56.893484Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:56.899445Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.899445Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:7aad2f2858aa8f80ce0687a6ac81b45083c4eaf1b99ff406e5f9bc09c7f18824","observation_id":"30beaffb-ffef-47f2-8de3-7204ad86a871","resolution":{"observed_at":"2026-08-16T05:29:56.899445Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:56.905202Z","title":"Walter, and D","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.905202Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:fe389e75e11b598dea3e3e74dbc078299afb99302884f0d039904a24fc1a4b8d","observation_id":"06e21d68-1da0-48ce-bb7f-dafdf6cda99c","resolution":{"observed_at":"2026-08-16T05:29:56.905202Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:58.298241Z","title":"Levinson","venue":null,"work_id":"efb9ab39-8178-468e-accc-88a09891d682","year":2003},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.911589Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:1b12e8c8691b86b11e657b660f2e6f50cfd021c5770f944b2b3ec8ba1eaeb92d","observation_id":"26456c22-6c49-4846-8312-416b484a6b8f","resolution":{"observed_at":"2026-08-16T05:29:58.305535Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.05463","last_updated":"2023-09-11T14:01:45Z","snapshot_observed_at":"2026-08-02T22:47:03.212781Z","submitted_at":"2023-09-11T14:01:45Z","title":"Textbooks Are All You Need II: phi-1.5 technical report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.05463","snapshot_observed_at":"2026-08-16T05:29:56.917006Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.917006Z"},"links":{"cited_paper":"/paper/2309.05463","citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:f742c448fd6048d0231984baacfa5d72bcc4521258976f555aa88e83f1ea11d2","observation_id":"2f8bdac0-6976-4b02-9901-5c5c6dbc4664","resolution":{"observed_at":"2026-08-16T05:29:56.917006Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:56.923474Z","title":null,"venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.923474Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:1df224a4924f68f4b830cc11e1ad2d4515c04994d284c8d4bd4bafa20df3e3af","observation_id":"b81431a9-c4a1-4a66-9977-c489dfea70eb","resolution":{"observed_at":"2026-08-16T05:29:56.923474Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:56.929535Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.929535Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:f78fb894e38a6f15065c9c9026c7ba287cda1911a57bd8c5701955215af666c6","observation_id":"599eefb2-e2b0-4fa6-86cf-7ff3bd154741","resolution":{"observed_at":"2026-08-16T05:29:56.929535Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:58.212605Z","title":null,"venue":null,"work_id":"444bb893-1ff4-455b-abec-db7543b541b4","year":2023},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.936210Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:f66aa8598ac1206ca2430b81d3c52ff2fb96befc753a22b2a485bb6095311a6c","observation_id":"b7c815cc-06c2-47cd-8d2b-43467dd52b4e","resolution":{"observed_at":"2026-08-16T05:29:58.228568Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:58.182977Z","title":null,"venue":null,"work_id":"74ddb943-45b8-4f64-ae67-0b661226fb04","year":2024},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.941529Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:8e98e13b72289e9e109b795989c04b22ede047afb975483c4b21f2578ef370d8","observation_id":"dd90cef3-a258-4c4d-8f2d-b70ae563621e","resolution":{"observed_at":"2026-08-16T05:29:58.192453Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:56.947470Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.947470Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:d36481ed3a3149892a0ba560f7ed52de5a45b38d10abcca6b94e68ca8827cc62","observation_id":"87f2b336-66b5-4f01-878c-66b93308c02a","resolution":{"observed_at":"2026-08-16T05:29:56.947470Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:56.952819Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.952819Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:6a8a3a2f6904e35217595977c88ae7f6062f9e043e74623e229b32c97bee5d31","observation_id":"62abab0d-91e8-43f9-aee9-c705f773c547","resolution":{"observed_at":"2026-08-16T05:29:56.952819Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:56.959146Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.959146Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:581d9efda40a9b47a2447fbfef937eca6427d7ce7de7993d1f1250a9227a5ca8","observation_id":"7743a31a-7e5c-4aaf-b62b-0d268b213ef8","resolution":{"observed_at":"2026-08-16T05:29:56.959146Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:58.103010Z","title":"Newcombe, Janellen, Huttenlocher, I","venue":null,"work_id":"104ceb9f-f653-41fe-88ed-c6f331f8f6de","year":2000},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.965301Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:a1ec770940d71acea1aec6018b17ce1ec3ff6cef950f0bcba6ae93092f1d7bae","observation_id":"f83a87b9-3e1c-457b-a3bd-a8fbaa16834b","resolution":{"observed_at":"2026-08-16T05:29:58.113275Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:58.051297Z","title":null,"venue":null,"work_id":"b223eee8-2876-45d9-b88d-8961419e800f","year":2024},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.971327Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:8e332c44df3931ac74f4ba63cf4616dfd1ffc2676451b3fa0063b18df692db81","observation_id":"5574fa9b-c377-47a8-adb3-2d577a434ec7","resolution":{"observed_at":"2026-08-16T05:29:58.064682Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:58.025151Z","title":null,"venue":null,"work_id":"73dfd78e-da00-4b9d-9c75-7cfb3d8eb998","year":2020},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.977526Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:c376d7e62c5b3bc0dc05121b3cf2d4324550c88e780c280bda456b6150e7461e","observation_id":"ca85c258-a0d4-4b3e-a60f-6914151230db","resolution":{"observed_at":"2026-08-16T05:29:58.031105Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:56.985406Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.985406Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:6b64d54c95babe9396b05e5fbf23f1b86566425ef9c563270ff1d23c355aebbc","observation_id":"164f3e1d-b61f-4990-9663-2cf9d4e1158a","resolution":{"observed_at":"2026-08-16T05:29:56.985406Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:57.975788Z","title":null,"venue":null,"work_id":"db9effdc-cc32-4fa3-bcf9-58cbe9cd65af","year":2021},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.991447Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:6150dbc09f2ce14b8e0ddf18f3190310cffc0efb44af80369b6d7ffb3d81ba91","observation_id":"cd6f2ad2-ec91-4b53-bb42-100732c79ae2","resolution":{"observed_at":"2026-08-16T05:29:57.987739Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:57.943011Z","title":null,"venue":null,"work_id":"6367df2c-10c2-4a4f-b1f1-06dfd0c59ad2","year":2019},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:56.997813Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:497ff48427734e73873f36a2b47fc951b085f6c18b101d3e76f95851b08c25d6","observation_id":"3af12fa0-e463-45ed-aa33-2a92ed0c19cb","resolution":{"observed_at":"2026-08-16T05:29:57.950525Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:57.003287Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:57.003287Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:284ce99a14818d7bb31f6cc87c49fa6fe2c3bcc2d17025c890dbbb52ae05a82b","observation_id":"aa4f0c11-6f26-4b6d-b3c2-040298f1b3e2","resolution":{"observed_at":"2026-08-16T05:29:57.003287Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:57.009662Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:57.009662Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:6b6abf5f31f9213185fe89c20fabaa3b2096982fb0f1228a68d4352085fae171","observation_id":"f36df284-13cc-4928-9b24-619a4d26c09f","resolution":{"observed_at":"2026-08-16T05:29:57.009662Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12191","last_updated":"2024-10-03T15:54:49Z","snapshot_observed_at":"2026-08-06T05:35:29.109022Z","submitted_at":"2024-09-18T17:59:32Z","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12191","snapshot_observed_at":"2026-08-16T05:29:57.015721Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:57.015721Z"},"links":{"cited_paper":"/paper/2409.12191","citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:3cd13f6ffb1998df505577ccdbd9678f52312ba22f496a755de868a8d2e81c85","observation_id":"04f32d84-27e9-4a02-9ed8-5861552905e5","resolution":{"observed_at":"2026-08-16T05:29:57.015721Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-08-17T18:50:07.059564Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-16T05:29:57.022762Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:57.022762Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:7922f5aad26f5558d8a0ae9f32cd3cee30c6a02167099f541d90136231212206","observation_id":"296c00ac-dc12-456a-987f-837c0a550436","resolution":{"observed_at":"2026-08-16T05:29:57.022762Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.13549","last_updated":"2024-11-29T15:51:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-23T15:21:52Z","title":"A Survey on Multimodal Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.13549","snapshot_observed_at":"2026-08-16T05:29:57.028551Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:57.028551Z"},"links":{"cited_paper":"/paper/2306.13549","citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:d1aa09a0d6364b1ca52e55237d3cdcb047de3e7077e41c723186af1f51287314","observation_id":"ca38951a-def4-45fe-84d3-4448efadc8ee","resolution":{"observed_at":"2026-08-16T05:29:57.028551Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:57.034689Z","title":null,"venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:57.034689Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:cf1eaf168d8fa7d846c849209f6cf482313c4f188f1c80359174614f495355ab","observation_id":"746c23f0-370e-4642-aaa5-63db45cde5b5","resolution":{"observed_at":"2026-08-16T05:29:57.034689Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:57.040369Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:57.040369Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:c13bba102ca8be7120f589a0ce15931777c8ae7d6dbb012cf6909d25be2fa962","observation_id":"4c84325d-c5a3-4b1f-8526-082542a30ef9","resolution":{"observed_at":"2026-08-16T05:29:57.040369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.17385","last_updated":"2025-04-17T17:59:27Z","snapshot_observed_at":"2026-08-17T19:38:20.646693Z","submitted_at":"2024-10-22T19:39:15Z","title":"Do Vision-Language Models Represent Space and How? Evaluating Spatial Frame of Reference Under Ambiguities","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.17385","snapshot_observed_at":"2026-08-16T05:29:57.046039Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:57.046039Z"},"links":{"cited_paper":"/paper/2410.17385","citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:ba31891a212c3e15ce993515168fdf35d3149f3198c5f3202942e19f5596bf34","observation_id":"c6ecd8fb-e5d4-45e4-a9ae-ae54aa22204e","resolution":{"observed_at":"2026-08-16T05:29:57.046039Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:57.826734Z","title":null,"venue":null,"work_id":"a6fe2ed5-b5c2-4f02-873a-17d80e182012","year":2019},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:57.054082Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:977307e5c698cab11a3ef485fa9eb4befbb67c2439dd78faecf2ee791c9c3153","observation_id":"79e5acdb-eb8a-4dbe-9ac6-79ddc6834caf","resolution":{"observed_at":"2026-08-16T05:29:57.833475Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:57.059772Z","title":"online\" 'onlinestring :=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:57.059772Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:7ed95d4fc0156c5e2abdbc5fd1f4cd9de0350667572501fc385bfa9781b37894","observation_id":"b6bec2ed-466c-4631-a9f0-2b8c2b5b8376","resolution":{"observed_at":"2026-08-16T05:29:57.059772Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T05:29:57.069202Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:57.069202Z"},"links":{"citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:bdce8432d89f220772d24ff525f6752bdd434c7bf4a87b6991d4950c80f0b687","observation_id":"28021e7a-dffa-48c0-b245-df0471f87156","resolution":{"observed_at":"2026-08-16T05:29:57.069202Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data"},"reference_resolution":{"displayed":49,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":45,"verified_exact":1,"verified_fuzzy":3},"total_outbound_references":49},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"thesis":"As of 21 August 2026, this Paper Citation Record lists 49 of 49 outbound references and 3 inbound Pith citation observations for arXiv:2504.20648."}