{"as_of":"2026-08-15T07:09:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e0730900a9531bf778d5ef19dc1e21c24c5ca5869afd81cdf49a1205502470ea","coverage":[{"denominator":27,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":27,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T10:29:05.612175Z","state":"measured"},{"denominator":27,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":27,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-15T06:32:42.880941+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2412.16599/citation-record","integrity":"/paper/2412.16599/integrity","json":"/paper/2412.16599/citation-record.json","paper":"/paper/2412.16599"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"1803.10122","last_updated":"2018-05-09T09:06:27Z","snapshot_observed_at":"2026-07-31T21:36:45.596575Z","submitted_at":"2018-03-27T15:08:55Z","title":"World Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1803.10122","snapshot_observed_at":"2026-08-11T10:29:05.487011Z","title":"World models,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2412.16599","last_updated":"2024-12-21T12:09:13Z","snapshot_observed_at":"2026-08-14T03:32:31.809362Z","submitted_at":"2024-12-21T12:09:13Z","title":"Do Multimodal Language Models Really Understand Direction? A Benchmark for Compass Direction Reasoning","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-11T10:29:05.487011Z"},"links":{"cited_paper":"/paper/1803.10122","citing_paper":"/paper/2412.16599"},"observation_digest":"sha256:5e9eb30ea3139bc677dd237cb1215ea0198a4fc36cecde699dfc8932f21b46eb","observation_id":"86650132-cfd7-4c3e-ba1c-24f31881fd7f","resolution":{"observed_at":"2026-08-11T10:29:05.487011Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.14992","last_updated":"2023-10-23T07:24:28Z","snapshot_observed_at":"2026-07-06T15:32:25.931739Z","submitted_at":"2023-05-24T10:28:28Z","title":"Reasoning with Language Model is Planning with World Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.14992","snapshot_observed_at":"2026-08-11T10:29:05.493027Z","title":"Reasoning with language model is planning with world model,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.16599","last_updated":"2024-12-21T12:09:13Z","snapshot_observed_at":"2026-08-14T03:32:31.809362Z","submitted_at":"2024-12-21T12:09:13Z","title":"Do Multimodal Language Models Really Understand Direction? A Benchmark for Compass Direction Reasoning","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-11T10:29:05.493027Z"},"links":{"cited_paper":"/paper/2305.14992","citing_paper":"/paper/2412.16599"},"observation_digest":"sha256:45c351fb7f0e0e2dd60bae9020a922bc37d79a2aa59a448755768c65447ace8a","observation_id":"e1de751e-d817-4cbe-8b07-792ac818aa0e","resolution":{"observed_at":"2026-08-11T10:29:05.493027Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.08268","last_updated":"2025-02-03T21:47:31Z","snapshot_observed_at":"2026-08-14T09:16:47.522119Z","submitted_at":"2024-02-13T07:47:36Z","title":"World Model on Million-Length Video And Language With Blockwise RingAttention","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.08268","snapshot_observed_at":"2026-08-11T10:29:05.498538Z","title":"World model on million-length video and language with ringattention,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.16599","last_updated":"2024-12-21T12:09:13Z","snapshot_observed_at":"2026-08-14T03:32:31.809362Z","submitted_at":"2024-12-21T12:09:13Z","title":"Do Multimodal Language Models Really Understand Direction? A Benchmark for Compass Direction Reasoning","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-11T10:29:05.498538Z"},"links":{"cited_paper":"/paper/2402.08268","citing_paper":"/paper/2412.16599"},"observation_digest":"sha256:248d770d49a583754da96652d9ecc9638c46d9ea0883c1c93ded93f9a6fd3b7b","observation_id":"06652ee1-ac18-40bc-b991-4db7b0a3fdb3","resolution":{"observed_at":"2026-08-11T10:29:05.498538Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T10:29:06.071656Z","title":"Penetrative ai: Making llms comprehend the physical world,","venue":null,"work_id":"ae7141d3-6103-4bb4-8061-bafe8a06b1aa","year":2024},"citing_paper":{"arxiv_id":"2412.16599","last_updated":"2024-12-21T12:09:13Z","snapshot_observed_at":"2026-08-14T03:32:31.809362Z","submitted_at":"2024-12-21T12:09:13Z","title":"Do Multimodal Language Models Really Understand Direction? A Benchmark for Compass Direction Reasoning","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-11T10:29:05.504412Z"},"links":{"citing_paper":"/paper/2412.16599"},"observation_digest":"sha256:37977f7762d03f1eb85018cba7bcb1fbc2a8454b62a93a7cde63307a025ba150","observation_id":"b9d84a79-efa2-4d21-9d46-5e40b1c062fc","resolution":{"observed_at":"2026-08-11T10:29:06.076764Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T10:29:06.055655Z","title":"Stepgame: A new benchmark for robust multi-hop spatial reasoning in texts,","venue":null,"work_id":"e8c5a44a-92d4-41ec-b3e7-6692daec887b","year":2022},"citing_paper":{"arxiv_id":"2412.16599","last_updated":"2024-12-21T12:09:13Z","snapshot_observed_at":"2026-08-14T03:32:31.809362Z","submitted_at":"2024-12-21T12:09:13Z","title":"Do Multimodal Language Models Really Understand Direction? A Benchmark for Compass Direction Reasoning","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-11T10:29:05.509538Z"},"links":{"citing_paper":"/paper/2412.16599"},"observation_digest":"sha256:6eae556d4b4e5326a3162f5a4106815ad4cbfca1a455a56d085a529499745525","observation_id":"263c16ad-82e0-4eb2-b2ba-d0f7479bb9a1","resolution":{"observed_at":"2026-08-11T10:29:06.060982Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T10:29:06.039944Z","title":"Eval- uating spatial understanding of large language models,","venue":null,"work_id":"ee8bb65e-49e9-4f0f-b7a4-16a96a8c50a7","year":2024},"citing_paper":{"arxiv_id":"2412.16599","last_updated":"2024-12-21T12:09:13Z","snapshot_observed_at":"2026-08-14T03:32:31.809362Z","submitted_at":"2024-12-21T12:09:13Z","title":"Do Multimodal Language Models Really Understand Direction? A Benchmark for Compass Direction Reasoning","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-11T10:29:05.514722Z"},"links":{"citing_paper":"/paper/2412.16599"},"observation_digest":"sha256:e74c01c95a6b6ca4180fbd95de48264e30976e028f878f1ee0377ad339cf3dbc","observation_id":"4496aaef-c2ed-474e-acb7-37ec9f096352","resolution":{"observed_at":"2026-08-11T10:29:06.045388Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.05832","last_updated":"2021-04-12T21:37:18Z","snapshot_observed_at":"2026-08-14T17:49:40.338667Z","submitted_at":"2021-04-12T21:37:18Z","title":"SpartQA: : A Textual Question Answering Benchmark for Spatial Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.05832","snapshot_observed_at":"2026-08-11T10:29:05.521033Z","title":"Spartqa:: A textual question answering benchmark for spatial reasoning,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.16599","last_updated":"2024-12-21T12:09:13Z","snapshot_observed_at":"2026-08-14T03:32:31.809362Z","submitted_at":"2024-12-21T12:09:13Z","title":"Do Multimodal Language Models Really Understand Direction? A Benchmark for Compass Direction Reasoning","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-11T10:29:05.521033Z"},"links":{"cited_paper":"/paper/2104.05832","citing_paper":"/paper/2412.16599"},"observation_digest":"sha256:8ad45c1cd707c179c96355cbd8c5fba678389b5c2067ec8a1c3bff6c10a8fdee","observation_id":"6b9fc6ba-4703-4fa6-a88d-dbcf55aad462","resolution":{"observed_at":"2026-08-11T10:29:05.521033Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03249","last_updated":"2025-02-24T00:58:13Z","snapshot_observed_at":"2026-08-13T05:57:09.255887Z","submitted_at":"2023-10-05T01:42:16Z","title":"Can Large Language Models be Good Path Planners? A Benchmark and Investigation on Spatial-temporal Reasoning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03249","snapshot_observed_at":"2026-08-11T10:29:05.526241Z","title":"Can large language models be good path planners? A benchmark and investigation on spatial-temporal reasoning,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.16599","last_updated":"2024-12-21T12:09:13Z","snapshot_observed_at":"2026-08-14T03:32:31.809362Z","submitted_at":"2024-12-21T12:09:13Z","title":"Do Multimodal Language Models Really Understand Direction? A Benchmark for Compass Direction Reasoning","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-11T10:29:05.526241Z"},"links":{"cited_paper":"/paper/2310.03249","citing_paper":"/paper/2412.16599"},"observation_digest":"sha256:62fa021f3eddaeb543f0be3338149ce6beda31b214267ab1d5fd0ba525d3558a","observation_id":"dee31865-f735-4b78-9d25-15fd47421547","resolution":{"observed_at":"2026-08-11T10:29:05.526241Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T10:29:06.021956Z","title":"Advancing spatial reasoning in large language models: An in-depth evaluation and enhancement using the stepgame benchmark,","venue":null,"work_id":"0ae063e2-ab5d-45fd-97b8-d2f9b17d0c88","year":2024},"citing_paper":{"arxiv_id":"2412.16599","last_updated":"2024-12-21T12:09:13Z","snapshot_observed_at":"2026-08-14T03:32:31.809362Z","submitted_at":"2024-12-21T12:09:13Z","title":"Do Multimodal Language Models Really Understand Direction? A Benchmark for Compass Direction Reasoning","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-11T10:29:05.531882Z"},"links":{"citing_paper":"/paper/2412.16599"},"observation_digest":"sha256:a05babe0d4ad74e5a4b621d3de25f198d6bcf653b8845f4ddbca96deff18f5b7","observation_id":"b6b1b0c3-c5a0-44a2-8877-cb0afababbe2","resolution":{"observed_at":"2026-08-11T10:29:06.028494Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-11T10:29:05.536601Z","title":"Gpt-4 technical report,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.16599","last_updated":"2024-12-21T12:09:13Z","snapshot_observed_at":"2026-08-14T03:32:31.809362Z","submitted_at":"2024-12-21T12:09:13Z","title":"Do Multimodal Language Models Really Understand Direction? A Benchmark for Compass Direction Reasoning","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-11T10:29:05.536601Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2412.16599"},"observation_digest":"sha256:5177c7d801f31ac62b154c17a8da09b027c17414e2413f891332206cd5864b86","observation_id":"b76c1a05-afa7-40b1-9724-21d0df1cc0e3","resolution":{"observed_at":"2026-08-11T10:29:05.536601Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T10:29:05.541338Z","title":"Learning transferable visual models from natural language supervision,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.16599","last_updated":"2024-12-21T12:09:13Z","snapshot_observed_at":"2026-08-14T03:32:31.809362Z","submitted_at":"2024-12-21T12:09:13Z","title":"Do Multimodal Language Models Really Understand Direction? A Benchmark for Compass Direction Reasoning","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-11T10:29:05.541338Z"},"links":{"citing_paper":"/paper/2412.16599"},"observation_digest":"sha256:8aa47d75cc78cbb1cee9c03fd05a543361ecd3ff7b9910453797a685cb889ef5","observation_id":"634969cb-f05d-4736-ac92-1799865ddd12","resolution":{"observed_at":"2026-08-11T10:29:05.541338Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T10:29:05.991874Z","title":"Scaling up vision-language pre-training for image captioning,","venue":null,"work_id":"6853aa61-b504-4e34-97f5-8a4fe4a2ec76","year":2022},"citing_paper":{"arxiv_id":"2412.16599","last_updated":"2024-12-21T12:09:13Z","snapshot_observed_at":"2026-08-14T03:32:31.809362Z","submitted_at":"2024-12-21T12:09:13Z","title":"Do Multimodal Language Models Really Understand Direction? A Benchmark for Compass Direction Reasoning","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-11T10:29:05.546151Z"},"links":{"citing_paper":"/paper/2412.16599"},"observation_digest":"sha256:de469086a11f5ee572cb7a33473badc1105dad25a79dc8277add50ce118755ea","observation_id":"4fb1afa3-aee2-40c9-b75a-972bd31ae5b9","resolution":{"observed_at":"2026-08-11T10:29:05.997392Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T10:29:05.975318Z","title":"Visual instruction tuning,","venue":null,"work_id":"314ec185-c93a-42b0-8baf-c2a4a95302b0","year":2023},"citing_paper":{"arxiv_id":"2412.16599","last_updated":"2024-12-21T12:09:13Z","snapshot_observed_at":"2026-08-14T03:32:31.809362Z","submitted_at":"2024-12-21T12:09:13Z","title":"Do Multimodal Language Models Really Understand Direction? A Benchmark for Compass Direction Reasoning","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-11T10:29:05.550294Z"},"links":{"citing_paper":"/paper/2412.16599"},"observation_digest":"sha256:0d9de4055cb7f22594bb0370c75e5404e85e3ed10183fb1183d3c2a0b018b459","observation_id":"ee1d367d-2c0c-49b0-8369-e01b9a11d65c","resolution":{"observed_at":"2026-08-11T10:29:05.980749Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T10:29:05.958054Z","title":"MODE: a multimodal open-domain dialogue dataset with explanation,","venue":null,"work_id":"b37bd1fa-a055-45d2-8175-5a86e4926296","year":2024},"citing_paper":{"arxiv_id":"2412.16599","last_updated":"2024-12-21T12:09:13Z","snapshot_observed_at":"2026-08-14T03:32:31.809362Z","submitted_at":"2024-12-21T12:09:13Z","title":"Do Multimodal Language Models Really Understand Direction? A Benchmark for Compass Direction Reasoning","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-11T10:29:05.554412Z"},"links":{"citing_paper":"/paper/2412.16599"},"observation_digest":"sha256:c0a3517fa4e7e36e878665547749f2588a6242ecf51c19395196b3367ec1638c","observation_id":"4da937dd-f47f-441b-b5dd-a729538d505d","resolution":{"observed_at":"2026-08-11T10:29:05.963609Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T10:29:05.940634Z","title":"Canny, The complexity of robot motion planning","venue":null,"work_id":"e7389821-6c9d-44a9-9d63-b6d0e21481a1","year":1988},"citing_paper":{"arxiv_id":"2412.16599","last_updated":"2024-12-21T12:09:13Z","snapshot_observed_at":"2026-08-14T03:32:31.809362Z","submitted_at":"2024-12-21T12:09:13Z","title":"Do Multimodal Language Models Really Understand Direction? A Benchmark for Compass Direction Reasoning","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-11T10:29:05.558686Z"},"links":{"citing_paper":"/paper/2412.16599"},"observation_digest":"sha256:ce8a5a0c589d533560d524d5b58f8be049970ad87490f921f416ee566ca03ce1","observation_id":"cdceea7b-3961-4914-9e6a-af78bac8385a","resolution":{"observed_at":"2026-08-11T10:29:05.946630Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T10:29:05.562695Z","title":null,"venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2412.16599","last_updated":"2024-12-21T12:09:13Z","snapshot_observed_at":"2026-08-14T03:32:31.809362Z","submitted_at":"2024-12-21T12:09:13Z","title":"Do Multimodal Language Models Really Understand Direction? A Benchmark for Compass Direction Reasoning","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-11T10:29:05.562695Z"},"links":{"citing_paper":"/paper/2412.16599"},"observation_digest":"sha256:5f98f3f13d6ffe8c705bf6c8c50014250cccae005f4afb9e3c8cf227c6341820","observation_id":"d9e5c415-fc7b-4cc8-9539-f16119af22f1","resolution":{"observed_at":"2026-08-11T10:29:05.562695Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T10:29:05.911435Z","title":"Intelligent control and decision- making demonstrated on a simple compass-guided robot,","venue":null,"work_id":"8a0a4c6f-cfd1-401b-a24c-17a10986c775","year":2000},"citing_paper":{"arxiv_id":"2412.16599","last_updated":"2024-12-21T12:09:13Z","snapshot_observed_at":"2026-08-14T03:32:31.809362Z","submitted_at":"2024-12-21T12:09:13Z","title":"Do Multimodal Language Models Really Understand Direction? A Benchmark for Compass Direction Reasoning","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-11T10:29:05.566617Z"},"links":{"citing_paper":"/paper/2412.16599"},"observation_digest":"sha256:00e18a10b9356de61c07f9ebac778553694209312941ddef6b3c2e01ecdbac8d","observation_id":"930a76a6-6306-4e9c-a06c-9835a40853af","resolution":{"observed_at":"2026-08-11T10:29:05.916989Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T10:29:05.893943Z","title":"Plasticity of human spatial cognition: Spatial language and cognition covary across cultures,","venue":null,"work_id":"727af60b-c070-456a-a3a6-fffd20b9b551","year":2011},"citing_paper":{"arxiv_id":"2412.16599","last_updated":"2024-12-21T12:09:13Z","snapshot_observed_at":"2026-08-14T03:32:31.809362Z","submitted_at":"2024-12-21T12:09:13Z","title":"Do Multimodal Language Models Really Understand Direction? A Benchmark for Compass Direction Reasoning","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-11T10:29:05.570617Z"},"links":{"citing_paper":"/paper/2412.16599"},"observation_digest":"sha256:1418e76d3bdf76efe5ca398e6618899c6c1c5fdf20f31604dec503b173f64c5a","observation_id":"b1884237-a6bb-45f9-850a-40c0cbe2aa08","resolution":{"observed_at":"2026-08-11T10:29:05.899360Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T10:29:05.878264Z","title":null,"venue":null,"work_id":"4d608e24-c3d4-470c-8e74-1040396e89f0","year":2003},"citing_paper":{"arxiv_id":"2412.16599","last_updated":"2024-12-21T12:09:13Z","snapshot_observed_at":"2026-08-14T03:32:31.809362Z","submitted_at":"2024-12-21T12:09:13Z","title":"Do Multimodal Language Models Really Understand Direction? A Benchmark for Compass Direction Reasoning","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-11T10:29:05.574525Z"},"links":{"citing_paper":"/paper/2412.16599"},"observation_digest":"sha256:bf106d1c80e9c7e8e6934196ddc986c6f6e4e9c2ff642ac7ed8f68914dfd8adc","observation_id":"abb1f37a-3880-4ab1-8888-6c32844914e1","resolution":{"observed_at":"2026-08-11T10:29:05.883313Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03622","last_updated":"2024-10-23T07:20:26Z","snapshot_observed_at":"2026-08-13T00:37:32.109106Z","submitted_at":"2024-04-04T17:45:08Z","title":"Mind's Eye of LLMs: Visualization-of-Thought Elicits Spatial Reasoning in Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.03622","snapshot_observed_at":"2026-08-11T10:29:05.578679Z","title":"Mind’s eye of llms: Visualization-of-thought elicits spatial reasoning in large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.16599","last_updated":"2024-12-21T12:09:13Z","snapshot_observed_at":"2026-08-14T03:32:31.809362Z","submitted_at":"2024-12-21T12:09:13Z","title":"Do Multimodal Language Models Really Understand Direction? A Benchmark for Compass Direction Reasoning","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-11T10:29:05.578679Z"},"links":{"cited_paper":"/paper/2404.03622","citing_paper":"/paper/2412.16599"},"observation_digest":"sha256:1e9e0f8b82d72722fcc44157086ea68ee8fce7075afc1cf7454f84cbe58b8393","observation_id":"9e875c7d-0722-46b8-aaf1-4a8b6bb602b9","resolution":{"observed_at":"2026-08-11T10:29:05.578679Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T10:29:05.861344Z","title":"Visual spatial reasoning,","venue":null,"work_id":"b6a53503-dfee-45f5-be44-49bf5f05f3a4","year":2023},"citing_paper":{"arxiv_id":"2412.16599","last_updated":"2024-12-21T12:09:13Z","snapshot_observed_at":"2026-08-14T03:32:31.809362Z","submitted_at":"2024-12-21T12:09:13Z","title":"Do Multimodal Language Models Really Understand Direction? A Benchmark for Compass Direction Reasoning","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-11T10:29:05.583011Z"},"links":{"citing_paper":"/paper/2412.16599"},"observation_digest":"sha256:ced9d2ffc0d7535207aae79cde840343e38845153d8e75f451e93fe1ad4c919c","observation_id":"e5a606e8-74fe-4f05-bc13-4f53f5c4652a","resolution":{"observed_at":"2026-08-11T10:29:05.866361Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.19785","last_updated":"2023-10-30T17:50:15Z","snapshot_observed_at":"2026-08-14T23:43:55.212577Z","submitted_at":"2023-10-30T17:50:15Z","title":"What's \"up\" with vision-language models? Investigating their struggle with spatial reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.19785","snapshot_observed_at":"2026-08-11T10:29:05.587498Z","title":"What’s” up","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.16599","last_updated":"2024-12-21T12:09:13Z","snapshot_observed_at":"2026-08-14T03:32:31.809362Z","submitted_at":"2024-12-21T12:09:13Z","title":"Do Multimodal Language Models Really Understand Direction? A Benchmark for Compass Direction Reasoning","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-11T10:29:05.587498Z"},"links":{"cited_paper":"/paper/2310.19785","citing_paper":"/paper/2412.16599"},"observation_digest":"sha256:9f6f6d9b8011a2686481f217f171f8e9cfc243437f9d5c38747e52432ce2c107","observation_id":"557c7b6f-1b2d-44cc-86b2-e53dbf62dc5f","resolution":{"observed_at":"2026-08-11T10:29:05.587498Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T10:29:05.844773Z","title":"Improved baselines with visual instruction tuning,","venue":null,"work_id":"98533501-3fc8-49f1-b91a-de27d7317c94","year":2024},"citing_paper":{"arxiv_id":"2412.16599","last_updated":"2024-12-21T12:09:13Z","snapshot_observed_at":"2026-08-14T03:32:31.809362Z","submitted_at":"2024-12-21T12:09:13Z","title":"Do Multimodal Language Models Really Understand Direction? A Benchmark for Compass Direction Reasoning","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-11T10:29:05.592077Z"},"links":{"citing_paper":"/paper/2412.16599"},"observation_digest":"sha256:aa9134ee124c03a7eb69122db0dd8863eeabc9f611edb8784b84bd365db24337","observation_id":"e8b71fba-e78b-4a87-828b-981acb1d03ae","resolution":{"observed_at":"2026-08-11T10:29:05.850085Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T10:29:05.597085Z","title":"The claude 3 model family: Opus, sonnet, haiku,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.16599","last_updated":"2024-12-21T12:09:13Z","snapshot_observed_at":"2026-08-14T03:32:31.809362Z","submitted_at":"2024-12-21T12:09:13Z","title":"Do Multimodal Language Models Really Understand Direction? A Benchmark for Compass Direction Reasoning","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-11T10:29:05.597085Z"},"links":{"citing_paper":"/paper/2412.16599"},"observation_digest":"sha256:deed6a49b43f0c5259f143ef97e3585c29e018c8670d559642e647d00a9b518f","observation_id":"12ee3148-c720-496e-b595-1eeb9305831c","resolution":{"observed_at":"2026-08-11T10:29:05.597085Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T10:29:05.815169Z","title":"Gpt-4o mini: A smaller, cheaper ai model,","venue":null,"work_id":"9dda9466-9373-4d36-9a1a-2958a1611c63","year":2024},"citing_paper":{"arxiv_id":"2412.16599","last_updated":"2024-12-21T12:09:13Z","snapshot_observed_at":"2026-08-14T03:32:31.809362Z","submitted_at":"2024-12-21T12:09:13Z","title":"Do Multimodal Language Models Really Understand Direction? A Benchmark for Compass Direction Reasoning","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-11T10:29:05.601779Z"},"links":{"citing_paper":"/paper/2412.16599"},"observation_digest":"sha256:a10ff8d0f437968badb8fb2b77a8b175065d4c9d767ff05bf8d8071efb456372","observation_id":"3b5a6f9e-fe49-470e-99c8-cac06d98752d","resolution":{"observed_at":"2026-08-11T10:29:05.822641Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05530","last_updated":"2024-12-16T17:39:39Z","snapshot_observed_at":"2026-08-14T18:15:53.516440Z","submitted_at":"2024-03-08T18:54:20Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05530","snapshot_observed_at":"2026-08-11T10:29:05.606668Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.16599","last_updated":"2024-12-21T12:09:13Z","snapshot_observed_at":"2026-08-14T03:32:31.809362Z","submitted_at":"2024-12-21T12:09:13Z","title":"Do Multimodal Language Models Really Understand Direction? A Benchmark for Compass Direction Reasoning","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-11T10:29:05.606668Z"},"links":{"cited_paper":"/paper/2403.05530","citing_paper":"/paper/2412.16599"},"observation_digest":"sha256:9e3f2332a2140373e69bfd952230365420f750e4cd2f859166e3eda09e51a134","observation_id":"447d4899-bc7a-4edc-9c37-ffd815ad3125","resolution":{"observed_at":"2026-08-11T10:29:05.606668Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T10:29:05.796257Z","title":"Ali iconfont,","venue":null,"work_id":"c2d9abce-1448-4751-bb2e-3624b8a790ec","year":2024},"citing_paper":{"arxiv_id":"2412.16599","last_updated":"2024-12-21T12:09:13Z","snapshot_observed_at":"2026-08-14T03:32:31.809362Z","submitted_at":"2024-12-21T12:09:13Z","title":"Do Multimodal Language Models Really Understand Direction? A Benchmark for Compass Direction Reasoning","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-11T10:29:05.612175Z"},"links":{"citing_paper":"/paper/2412.16599"},"observation_digest":"sha256:26fc9a19304c08750104a357a17d4898aaf77644587eee1d1095abfa6dbf5a47","observation_id":"683d4dd7-22e4-4a3b-8873-1ad1e09eeae8","resolution":{"observed_at":"2026-08-11T10:29:05.802894Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2412.16599","last_updated":"2024-12-21T12:09:13Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-14T03:32:31.809362Z","submitted_at":"2024-12-21T12:09:13Z","title":"Do Multimodal Language Models Really Understand Direction? A Benchmark for Compass Direction Reasoning"},"reference_resolution":{"displayed":27,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":13,"verified_exact":0,"verified_fuzzy":14},"total_outbound_references":27},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"thesis":"As of 15 August 2026, this Paper Citation Record lists 27 of 27 outbound references and 0 inbound Pith citation observations for arXiv:2412.16599."}