{"as_of":"2026-08-10T10:12:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:1b83102aa9ab1bb6ff7a11ebac0cfc25535d3d94fb7311bf7b48c46aca6a90f4","coverage":[{"denominator":21,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":21,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T13:40:02.915482Z","state":"measured"},{"denominator":26,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":26,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":5,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":5,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T13:39:59.590925Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T08:59:43.323576Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-08T23:51:41.898919Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.20395","snapshot_observed_at":"2026-08-06T13:39:59.590925Z","title":"As LLMs increasingly serve as the foundation for autonomous agents (Duan et al., 2022), understanding their capacity for spatial rea- soningbecomescrucial","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-08T23:51:41.898919Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T13:39:59.590925Z"},"links":{"cited_paper":"/paper/2507.20395","citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:53f714b9b60939c150c6bd9a13c3f1e6fd0c8d0074563e4bdf76df1ad1235054","observation_id":"e40e9f1c-26bb-4f59-a470-b31aae587614","resolution":{"observed_at":"2026-08-06T13:39:59.590925Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-08T23:51:41.898919Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"cited_work":{"arxiv_id":"2507.20395","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.20395","snapshot_observed_at":"2026-07-04T08:59:43.323576Z","title":"Mazeeval: A benchmark for testing sequential decision-making in language models","venue":null,"work_id":"529e0c95-b668-41c6-a159-f245a0c5365e","year":2025},"citing_paper":{"arxiv_id":"2605.09965","last_updated":"2026-05-12T15:54:46Z","snapshot_observed_at":"2026-07-06T23:21:59.096464Z","submitted_at":"2026-05-11T04:16:41Z","title":"Towards Generalist Game Players: An Investigation of Foundation Models in the Game Multiverse","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-12T03:25:24.844859Z"},"links":{"cited_paper":"/paper/2507.20395","citing_paper":"/paper/2605.09965"},"observation_digest":"sha256:0ed41d691ccbf8c5b5928a1ca5f6688197db5d16845b0da060eb34024512abf7","observation_id":"2bfd5544-d885-4fae-a784-63ccd80d8a5a","resolution":{"observed_at":"2026-05-12T03:26:18.991223Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-08T23:51:41.898919Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"cited_work":{"arxiv_id":"2507.20395","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.20395","snapshot_observed_at":"2026-07-04T08:59:43.323576Z","title":"Mazeeval: A benchmark for testing sequential decision-making in language models","venue":null,"work_id":"529e0c95-b668-41c6-a159-f245a0c5365e","year":2025},"citing_paper":{"arxiv_id":"2605.09965","last_updated":"2026-05-12T15:54:46Z","snapshot_observed_at":"2026-07-06T23:21:59.096464Z","submitted_at":"2026-05-11T04:16:41Z","title":"Towards Generalist Game Players: An Investigation of Foundation Models in the Game Multiverse","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-13T06:44:28.552513Z"},"links":{"cited_paper":"/paper/2507.20395","citing_paper":"/paper/2605.09965"},"observation_digest":"sha256:bd92f83a84eb6666fca59fa8ae8a8255e090220208f5ed6da2151d11c2a93bb8","observation_id":"8a1c1c7b-fa67-4df1-a8dd-2da3ab196fa4","resolution":{"observed_at":"2026-05-13T06:47:27.126726Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-08T23:51:41.898919Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"cited_work":{"arxiv_id":"2507.20395","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.20395","snapshot_observed_at":"2026-07-04T08:59:43.323576Z","title":"Mazeeval: A benchmark for testing sequential decision-making in language models","venue":null,"work_id":"529e0c95-b668-41c6-a159-f245a0c5365e","year":2025},"citing_paper":{"arxiv_id":"2605.28277","last_updated":"2026-05-27T10:20:53Z","snapshot_observed_at":"2026-08-01T19:48:22.299809Z","submitted_at":"2026-05-27T10:20:53Z","title":"Do LLMs Build World Models From Text? A Multilingual Diagnostic of Spatial Reasoning","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-29T12:02:14.499758Z"},"links":{"cited_paper":"/paper/2507.20395","citing_paper":"/paper/2605.28277"},"observation_digest":"sha256:f6fb5c980ea2a50b3b62d57a0485ad9245c8845808a47cb6d1b59b85d5cee728","observation_id":"b3db04d6-9c50-4606-b4a3-b4e5bc6bb440","resolution":{"observed_at":"2026-06-29T12:03:23.712055Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-08T23:51:41.898919Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"cited_work":{"arxiv_id":"2507.20395","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.20395","snapshot_observed_at":"2026-07-04T08:59:43.323576Z","title":"Mazeeval: A benchmark for testing sequential decision-making in language models","venue":null,"work_id":"529e0c95-b668-41c6-a159-f245a0c5365e","year":2025},"citing_paper":{"arxiv_id":"2606.22219","last_updated":"2026-06-20T20:41:43Z","snapshot_observed_at":"2026-08-08T21:42:34.105257Z","submitted_at":"2026-06-20T20:41:43Z","title":"Lost in Aggregation: A Multi-Scale Diagnostic Benchmark for LLM Spatial Navigation","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-26T10:37:24.946718Z"},"links":{"cited_paper":"/paper/2507.20395","citing_paper":"/paper/2606.22219"},"observation_digest":"sha256:daab5f889176b116847cd9f3074a8826fded3f24a9ff559eeda20865a9ed3f65","observation_id":"c66416b3-7501-44d1-9685-c2c624d4d082","resolution":{"observed_at":"2026-07-04T08:59:43.325606Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2507.20395/citation-record","integrity":"/paper/2507.20395/integrity","json":"/paper/2507.20395/citation-record.json","paper":"/paper/2507.20395"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-08T23:51:41.898919Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.20395","snapshot_observed_at":"2026-08-06T13:39:59.590925Z","title":"As LLMs increasingly serve as the foundation for autonomous agents (Duan et al., 2022), understanding their capacity for spatial rea- soningbecomescrucial","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-08T23:51:41.898919Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T13:39:59.590925Z"},"links":{"cited_paper":"/paper/2507.20395","citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:53f714b9b60939c150c6bd9a13c3f1e6fd0c8d0074563e4bdf76df1ad1235054","observation_id":"e40e9f1c-26bb-4f59-a470-b31aae587614","resolution":{"observed_at":"2026-08-06T13:39:59.590925Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:40:09.464754Z","title":null,"venue":null,"work_id":"2e43031b-22f5-4cdb-93c3-97f2806bf265","year":2022},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-08T23:51:41.898919Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T13:39:59.694094Z"},"links":{"citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:445bb2b5ca7537c9433a0907a680d1573fa67cd7370310942e8c988880009de7","observation_id":"99541be7-f270-4b91-aed6-9cbdbb8bbbcc","resolution":{"observed_at":"2026-08-06T13:40:09.594740Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:40:08.350782Z","title":"We focus on the configurationthatprovidesthemostchallengingyet fair assessment of spatial reasoning capabilities","venue":null,"work_id":"8469e64b-1557-4aa2-b0f4-4b8c119a6d3a","year":2023},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-08T23:51:41.898919Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T13:40:00.204151Z"},"links":{"citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:d195f00914b4df303e4fc210ffee5a011e42fe933d3d21a1a4fa58f4ded40a56","observation_id":"0d21d79d-f7d3-4b02-82d3-c505c491a83e","resolution":{"observed_at":"2026-08-06T13:40:08.444915Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:40:08.086542Z","title":"The results re- veal significant variations in spatial reasoning capa- bilities and provide insights into how these abilities transfer across languages","venue":null,"work_id":"2265777b-904f-415a-b4fb-76e53fcd066d","year":null},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-08T23:51:41.898919Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T13:40:00.343327Z"},"links":{"citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:1c8372e9550d1bf4b8842b7e5eea1e6eba21a752fab6357a5618e689c644ee8d","observation_id":"41bb9291-c22c-4056-b142-7a36ac46df0c","resolution":{"observed_at":"2026-08-06T13:40:08.254906Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:40:07.784836Z","title":"illusion of thinking","venue":null,"work_id":"d7ee4d11-74b3-4339-b333-4bf6045ed2a5","year":2024},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-08T23:51:41.898919Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T13:40:00.487347Z"},"links":{"citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:efab01f167ed59b1250d7dec13ad77d18371547420c839744744e27e5164cc97","observation_id":"434ccbdb-126d-429c-b277-edef3801563a","resolution":{"observed_at":"2026-08-06T13:40:07.924848Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:40:06.985571Z","title":null,"venue":null,"work_id":"78aeacdc-cb93-4e5b-a5dc-3ac392f99e86","year":null},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-08T23:51:41.898919Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T13:40:00.765301Z"},"links":{"citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:fb24cff2f570f8561646298c8cadd8f5c6034123c2adcae02f2330c1feb52a6f","observation_id":"f50341c9-d0b7-4d17-ac02-04a6bfeb0dda","resolution":{"observed_at":"2026-08-06T13:40:07.194752Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:40:06.484759Z","title":null,"venue":null,"work_id":"87df8bf0-82ba-4f78-90b9-7a26cd9c612c","year":2025},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-08T23:51:41.898919Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T13:40:00.958377Z"},"links":{"citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:d6d48010b8c7697f89ba976ed6c753d3f3bfc21662663187d07961b8c9a69bc7","observation_id":"36f47d57-c02c-48d9-9ad0-6130d5b6d901","resolution":{"observed_at":"2026-08-06T13:40:06.673377Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03249","last_updated":"2025-02-24T00:58:13Z","snapshot_observed_at":"2026-08-05T14:15:00.395755Z","submitted_at":"2023-10-05T01:42:16Z","title":"Can Large Language Models be Good Path Planners? A Benchmark and Investigation on Spatial-temporal Reasoning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03249","snapshot_observed_at":"2026-08-06T13:40:01.114830Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-08T23:51:41.898919Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T13:40:01.114830Z"},"links":{"cited_paper":"/paper/2310.03249","citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:920b327178ae4c60f08a4f27e96f83352c036d89c509eb06586719238c8129cb","observation_id":"c22597bd-79f8-4566-8333-8d41d4303dd3","resolution":{"observed_at":"2026-08-06T13:40:01.114830Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11164","last_updated":"2023-04-22T06:28:46Z","snapshot_observed_at":"2026-08-02T01:09:20.439387Z","submitted_at":"2023-04-22T06:28:46Z","title":"Dialectical language model evaluation: An initial appraisal of the commonsense spatial reasoning abilities of LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11164","snapshot_observed_at":"2026-08-06T13:40:01.425301Z","title":"Marc-Alexandre Côté, Akos Kádár, Xingdi Yuan, Ben Kybartas, Tavian Barnes, Emery Fine, James Moore, Matthew Hausknecht, Layla El Asri, Mahmoud Adada, et al","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-08T23:51:41.898919Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T13:40:01.425301Z"},"links":{"cited_paper":"/paper/2304.11164","citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:45bc0b82e48ab967e02209d898f02084a7dafc27a99553ba6170d884f3464653","observation_id":"17eaa702-14d6-46c0-b172-226e24bd879a","resolution":{"observed_at":"2026-08-06T13:40:01.425301Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.00530","last_updated":"2025-04-23T10:26:16Z","snapshot_observed_at":"2026-08-09T10:18:49.335383Z","submitted_at":"2023-11-01T14:08:56Z","title":"Advances in Embodied Navigation Using Large Language Models: A Survey","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.00530","snapshot_observed_at":"2026-08-06T13:40:01.594827Z","title":"In Proceedings of the AAAI Conference on Artificial Intelligence, volume 38, pages 18500–18507","venue":null,"work_id":null,"year":2011},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-08T23:51:41.898919Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T13:40:01.594827Z"},"links":{"cited_paper":"/paper/2311.00530","citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:80012b3c183799cf323e2708aa25d21eefe92eeadd4f8f335a1d4702233962b5","observation_id":"4d31b4ae-c2ec-4b68-988f-4c949a17cf66","resolution":{"observed_at":"2026-08-06T13:40:01.594827Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.01054","last_updated":"2023-12-02T07:41:46Z","snapshot_observed_at":"2026-08-09T19:37:02.511310Z","submitted_at":"2023-12-02T07:41:46Z","title":"Exploring and Improving the Spatial Reasoning Abilities of Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.01054","snapshot_observed_at":"2026-08-06T13:40:01.783017Z","title":"InProceedings of the IEEE conferenceoncomputervisionandpatternrecog- nition, pages 8494–8502","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-08T23:51:41.898919Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T13:40:01.783017Z"},"links":{"cited_paper":"/paper/2312.01054","citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:a4739c21d117caa9b1b2d0a601d5eeabd5415ea626ea91d9ee20692857d7f1ab","observation_id":"f5ecded3-8aff-4c95-9681-4f77893aac7c","resolution":{"observed_at":"2026-08-06T13:40:01.783017Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:40:06.084830Z","title":null,"venue":null,"work_id":"d32aa752-bca3-41c8-b473-f7ccd7915eea","year":null},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-08T23:51:41.898919Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T13:40:02.184839Z"},"links":{"citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:64d1f925e391d46d1bb0401bc5d171f98e740ae1dfc7426be18d211a20e263b4","observation_id":"1f793967-6f8e-417a-9031-75ebb38f585e","resolution":{"observed_at":"2026-08-06T13:40:06.244842Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:40:05.624951Z","title":null,"venue":null,"work_id":"15fb2b37-8e30-4fa4-a163-304089d58593","year":null},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-08T23:51:41.898919Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T13:40:02.384763Z"},"links":{"citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:98c88117cf5a107ef02a5932fe97feec7726fd68719da16f160ca176447a4069","observation_id":"fcea007a-56e2-4d04-8b0a-8875684c0d63","resolution":{"observed_at":"2026-08-06T13:40:05.837369Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:40:05.135116Z","title":null,"venue":null,"work_id":"16840ffb-737a-45ea-96f4-c471baf68ff8","year":null},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-08T23:51:41.898919Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T13:40:02.604877Z"},"links":{"citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:4724d40ef4ab699b3daf4e962458ae82b74b873f937741b1808604367d4ef98a","observation_id":"d0996122-b4b5-4084-ac0b-27e838eba294","resolution":{"observed_at":"2026-08-06T13:40:05.374835Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:40:04.777170Z","title":null,"venue":null,"work_id":"691830e9-57e2-4cd8-8e17-103cdc1fe666","year":null},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-08T23:51:41.898919Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T13:40:02.759852Z"},"links":{"citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:b59bb67469582338b365ce376c8a8fd287f4ef41c823c4e8bf6f508f9e26171d","observation_id":"23523d3c-1ce1-4d82-a7d2-8e4b7cfd036d","resolution":{"observed_at":"2026-08-06T13:40:04.934759Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:40:04.385016Z","title":"Choose a direction: north, south, east, or west","venue":null,"work_id":"f1980128-40e6-4484-9a33-63be50720dc6","year":null},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-08T23:51:41.898919Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T13:40:02.915482Z"},"links":{"citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:444eb9ddcdc7c2120ecefc9c1adf6418bcb1e874c175149e439c65189da8d3f0","observation_id":"7b9318bd-40fe-461a-8ce3-25d15b80fcc6","resolution":{"observed_at":"2026-08-06T13:40:04.577069Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1502.05698","last_updated":"2015-12-31T13:08:14Z","snapshot_observed_at":"2026-07-06T04:09:45.600447Z","submitted_at":"2015-02-19T20:46:10Z","title":"Towards AI-Complete Question Answering: A Set of Prerequisite Toy Tasks","version":10},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1502.05698","snapshot_observed_at":"2026-08-06T13:40:01.974813Z","title":"Jason Weston, Antoine Bordes, Sumit Chopra, Alexander M Rush, Bart Van Merriënboer, Ar- mand Joulin, and Tomas Mikolov","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-08T23:51:41.898919Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":2002,"source":"pdf_text","source_observed_at":"2026-08-06T13:40:01.974813Z"},"links":{"cited_paper":"/paper/1502.05698","citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:19a9fb29294f11783aefd3a11d0072a1b92f758b2533b30bdceb326fc7548d19","observation_id":"71c1483e-9315-4cdc-9085-a1149bcdf460","resolution":{"observed_at":"2026-08-06T13:40:01.974813Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:40:08.984850Z","title":"TextWorld (Côté et al., 2018) offers text-based navigation but in richly described environments that provide substantial contextual cues","venue":null,"work_id":"748f1abd-451e-4822-b777-4e6499d8c24f","year":2018},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-08T23:51:41.898919Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":2018,"source":"pdf_text","source_observed_at":"2026-08-06T13:39:59.837529Z"},"links":{"citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:09940af26cfa9c2af791f3634101f30c5493e3ec80ee0ea20865e458ab94ec60","observation_id":"c45dc4ac-9566-44a9-a228-fe49abb6f589","resolution":{"observed_at":"2026-08-06T13:40:09.194750Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:40:07.404863Z","title":null,"venue":null,"work_id":"565dbf0a-96a8-48fc-be23-bdf3434b6fea","year":2018},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-08T23:51:41.898919Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-06T13:40:00.634891Z"},"links":{"citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:05a1fc80ffcec746a8108e6fcb04262a19806abce97c5a3ce62d2ebcdcffb5f9","observation_id":"a9adacda-781c-4bf2-975b-d2bcac3bfe2a","resolution":{"observed_at":"2026-08-06T13:40:07.565032Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:40:08.640129Z","title":null,"venue":null,"work_id":"fc5240db-c6df-490f-ab4a-53e70e225cd4","year":null},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-08T23:51:41.898919Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-06T13:39:59.995856Z"},"links":{"citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:d909f669da30001d68b91555f65538105c5d59bf7254de8579f043d78fea6d0c","observation_id":"c2525298-d22f-4b43-ad67-4caf1691b83f","resolution":{"observed_at":"2026-08-06T13:40:08.774753Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1810.08272","last_updated":"2019-12-19T15:44:33Z","snapshot_observed_at":"2026-08-08T23:50:48.318949Z","submitted_at":"2018-10-18T20:48:08Z","title":"BabyAI: A Platform to Study the Sample Efficiency of Grounded Language Learning","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1810.08272","snapshot_observed_at":"2026-08-06T13:40:01.294846Z","title":"In Proceedings of the IEEE/CVF Conference on ComputerVisionandPatternRecognition, pages 14455–14465","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-08T23:51:41.898919Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-06T13:40:01.294846Z"},"links":{"cited_paper":"/paper/1810.08272","citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:27f2107882fac9b69cddd858d0691f1f008b3476757ef294874ade0fcecc71f9","observation_id":"fa740c9c-f980-4591-a60b-95ef6745e02f","resolution":{"observed_at":"2026-08-06T13:40:01.294846Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-08T23:51:41.898919Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models"},"reference_resolution":{"displayed":21,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":16,"verified_exact":0,"verified_fuzzy":5},"total_outbound_references":21},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 21 of 21 outbound references and 5 inbound Pith citation observations for arXiv:2507.20395."}