{"as_of":"2026-08-17T07:36:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:c0850dd885d64a23e94810ab9412f96f227bfc8f1b9b943ee5b0b3bbf2606f86","coverage":[{"denominator":73,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":73,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T11:42:30.874258Z","state":"measured"},{"denominator":74,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":74,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-17T06:30:58.91139+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T22:26:36.732238Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.14773","snapshot_observed_at":"2026-08-02T22:26:36.732238Z","title":"Planet: A collection of benchmarks for evaluating llms' planning capabilities","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.16902","last_updated":"2026-05-30T16:45:37Z","snapshot_observed_at":"2026-08-14T15:40:32.791963Z","submitted_at":"2026-02-18T21:33:59Z","title":"LLM-WikiRace Benchmark: How Far Can LLMs Plan over Real-World Knowledge Graphs?","version":4},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-02T22:26:36.732238Z"},"links":{"cited_paper":"/paper/2504.14773","citing_paper":"/paper/2602.16902"},"observation_digest":"sha256:4228de3062f83e74dcdf4101591ef37bd6baee63b6c16e178f2a8a9c95ec18b8","observation_id":"950f6c28-c90a-4404-93b6-ad96e4efce4c","resolution":{"observed_at":"2026-08-02T22:26:36.732238Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2504.14773/citation-record","integrity":"/paper/2504.14773/integrity","json":"/paper/2504.14773/citation-record.json","paper":"/paper/2504.14773"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:42:30.566702Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.566702Z"},"links":{"citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:dda0b045c972801fc0b04118c97fc5ae6343a8ff97d10e09bd714a22f9d938d9","observation_id":"437f4928-54b3-4d1b-bbf3-338c7770a7a8","resolution":{"observed_at":"2026-08-16T11:42:30.566702Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03249","last_updated":"2025-02-24T00:58:13Z","snapshot_observed_at":"2026-08-16T20:49:16.388563Z","submitted_at":"2023-10-05T01:42:16Z","title":"Can Large Language Models be Good Path Planners? A Benchmark and Investigation on Spatial-temporal Reasoning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03249","snapshot_observed_at":"2026-08-16T11:42:30.572023Z","title":"Can large language models be good path planners? a benchmark and investigation on spatial-temporal reasoning, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.572023Z"},"links":{"cited_paper":"/paper/2310.03249","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:3e292076d8b01bd79c4ba7995a08038c6a834c474e458d6293228cd71c05912f","observation_id":"800a59d7-ca20-4f31-bdac-1ff3fc66d388","resolution":{"observed_at":"2026-08-16T11:42:30.572023Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.05291","last_updated":"2025-02-05T21:50:07Z","snapshot_observed_at":"2026-08-16T13:36:28.398126Z","submitted_at":"2024-07-07T07:15:49Z","title":"WorkArena++: Towards Compositional Planning and Reasoning-based Common Knowledge Work Tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.05291","snapshot_observed_at":"2026-08-16T11:42:30.576526Z","title":"Workarena++: Towards compositional planning and reasoning-based common knowledge work tasks, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.576526Z"},"links":{"cited_paper":"/paper/2407.05291","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:c2987729322a1ae6bc02d8de03b267d7e0cda20674eca0eb8b33f427c1575271","observation_id":"f709a02c-cd01-4374-a0c9-91ac323d5e7f","resolution":{"observed_at":"2026-08-16T11:42:30.576526Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.08264","last_updated":"2024-09-13T20:17:13Z","snapshot_observed_at":"2026-08-16T13:19:02.749363Z","submitted_at":"2024-09-12T17:56:43Z","title":"Windows Agent Arena: Evaluating Multi-Modal OS Agents at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.08264","snapshot_observed_at":"2026-08-16T11:42:30.581189Z","title":"Windows agent arena: Evaluating multi-modal os agents at scale, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.581189Z"},"links":{"cited_paper":"/paper/2409.08264","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:7716aead247d06bef420b37b46eb27ecccdac110d31ce61293e7ee558938afaf","observation_id":"97692f16-a116-4764-a58a-102f8bf982aa","resolution":{"observed_at":"2026-08-16T11:42:30.581189Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.19472","last_updated":"2024-09-18T15:30:33Z","snapshot_observed_at":"2026-08-17T04:17:15.947850Z","submitted_at":"2023-05-31T00:55:40Z","title":"PlaSma: Making Small Language Models Better Procedural Knowledge Models for (Counterfactual) Planning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.19472","snapshot_observed_at":"2026-08-16T11:42:30.585566Z","title":"Hwang, Xiang Lorraine Li, Hirona J","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.585566Z"},"links":{"cited_paper":"/paper/2305.19472","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:3328396fdb8b783ab82472a49e04b04e04f86c60551191c67d5af76acda542fb","observation_id":"68bbc0d6-bcd5-4a8f-9e74-dffb0a2f6aed","resolution":{"observed_at":"2026-08-16T11:42:30.585566Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.00081","last_updated":"2024-10-31T17:53:12Z","snapshot_observed_at":"2026-08-16T13:04:02.599652Z","submitted_at":"2024-10-31T17:53:12Z","title":"PARTNR: A Benchmark for Planning and Reasoning in Embodied Multi-agent Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.00081","snapshot_observed_at":"2026-08-16T11:42:30.589774Z","title":"Turner, Eric Undersander, and Tsung-Yen Yang","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.589774Z"},"links":{"cited_paper":"/paper/2411.00081","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:b5dc16373183fdbf2ec34e14eeb39570b11018d306073a6d20488970b51e6cee","observation_id":"05665e30-b543-419c-8c3a-a6438eb26726","resolution":{"observed_at":"2026-08-16T11:42:30.589774Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.05746","last_updated":"2024-08-25T11:19:33Z","snapshot_observed_at":"2026-08-16T14:53:48.557440Z","submitted_at":"2023-10-09T14:22:09Z","title":"Put Your Money Where Your Mouth Is: Evaluating Strategic Planning and Execution of LLM Agents in an Auction Arena","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.05746","snapshot_observed_at":"2026-08-16T11:42:30.594239Z","title":"Put your money where your mouth is: Evaluating strategic planning and execution of llm agents in an auction arena, 2024 a","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.594239Z"},"links":{"cited_paper":"/paper/2310.05746","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:11090789f0c11cffcb84f219de468dfe888d768c1f6460367dc31c1225820c81","observation_id":"ed1af03c-e461-4663-a38e-530ff0dc4c76","resolution":{"observed_at":"2026-08-16T11:42:30.594239Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.06722","last_updated":"2024-06-11T06:53:44Z","snapshot_observed_at":"2026-08-16T19:41:21.903959Z","submitted_at":"2023-12-11T03:35:58Z","title":"EgoPlan-Bench: Benchmarking Multimodal Large Language Models for Human-Level Planning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.06722","snapshot_observed_at":"2026-08-16T11:42:30.598769Z","title":"Egoplan-bench: Benchmarking multimodal large language models for human-level planning, 2024 b","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.598769Z"},"links":{"cited_paper":"/paper/2312.06722","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:d186c3aee065b9a38ff3527f4607c2f152f8001e6f2fa491e589ad004858f159","observation_id":"b46779be-7e7d-4d18-9cdc-967121141d92","resolution":{"observed_at":"2026-08-16T11:42:30.598769Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.14762","last_updated":"2024-09-23T07:18:02Z","snapshot_observed_at":"2026-08-16T21:25:30.954659Z","submitted_at":"2024-09-23T07:18:02Z","title":"Do Large Language Models have Problem-Solving Capability under Incomplete Information Scenarios?","version":1},"cited_work":{"arxiv_id":"2409.14762","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.14762","snapshot_observed_at":"2026-08-16T11:42:31.748159Z","title":"Do Large Language Models have Problem-Solving Capability under Incomplete Information Scenarios?","venue":"cs.CL","work_id":"f1a3fa21-7a2f-439f-9df9-05676b609b37","year":2024},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.602664Z"},"links":{"cited_paper":"/paper/2409.14762","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:fc70c807edc62b6b182e027294ea8d6f4f58e04ce4dd92482e31e974f25c4c88","observation_id":"86f944a4-8e43-4c60-ae28-b7830bfac183","resolution":{"observed_at":"2026-08-16T11:42:31.753342Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.21033","last_updated":"2025-07-15T09:27:28Z","snapshot_observed_at":"2026-08-11T22:42:04.095207Z","submitted_at":"2024-12-30T15:58:41Z","title":"Plancraft: an evaluation dataset for planning with LLM agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.21033","snapshot_observed_at":"2026-08-16T11:42:30.606642Z","title":"Plancraft: an evaluation dataset for planning with llm agents, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.606642Z"},"links":{"cited_paper":"/paper/2412.21033","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:919f52eb08a8d800cb4188cf841e3736ffa19697bbe3cdc60d1e8808dab0b607","observation_id":"b7983657-5c1b-4998-817b-1220bf106969","resolution":{"observed_at":"2026-08-16T11:42:30.606642Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.06070","last_updated":"2023-12-09T05:57:46Z","snapshot_observed_at":"2026-08-14T08:19:18.935225Z","submitted_at":"2023-06-09T17:44:31Z","title":"Mind2Web: Towards a Generalist Agent for the Web","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.06070","snapshot_observed_at":"2026-08-16T11:42:30.610641Z","title":"Mind2web: Towards a generalist agent for the web, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.610641Z"},"links":{"cited_paper":"/paper/2306.06070","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:5b7f7071b53286a410956e0e1db37ee55184d53fa6b3bac71c02afacd11a43be","observation_id":"092da20e-953e-445f-8088-43b099aa739e","resolution":{"observed_at":"2026-08-16T11:42:30.610641Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.12348","last_updated":"2024-06-10T17:14:09Z","snapshot_observed_at":"2026-08-16T14:17:14.143141Z","submitted_at":"2024-02-19T18:23:36Z","title":"GTBench: Uncovering the Strategic Reasoning Limitations of LLMs via Game-Theoretic Evaluations","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.12348","snapshot_observed_at":"2026-08-16T11:42:30.614767Z","title":"Gtbench: Uncovering the strategic reasoning limitations of llms via game-theoretic evaluations, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.614767Z"},"links":{"cited_paper":"/paper/2402.12348","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:ff97d7cbbc8896fd59df5171cf99969ba9b25d54dff93d240194135d14a9d37f","observation_id":"04eed57b-eacd-4bb5-813e-9aa0ca54f0c9","resolution":{"observed_at":"2026-08-16T11:42:30.614767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.08853","last_updated":"2022-11-22T07:59:47Z","snapshot_observed_at":"2026-08-16T19:41:18.041989Z","submitted_at":"2022-06-17T15:53:05Z","title":"MineDojo: Building Open-Ended Embodied Agents with Internet-Scale Knowledge","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.08853","snapshot_observed_at":"2026-08-16T11:42:30.618903Z","title":"Minedojo: Building open-ended embodied agents with internet-scale knowledge, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.618903Z"},"links":{"cited_paper":"/paper/2206.08853","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:035a241b68a1749c53872d17946428357b629ba8f3096edb69ce8212bd6cae2f","observation_id":"fda76345-77e4-4320-b51e-1bcb93bba2bf","resolution":{"observed_at":"2026-08-16T11:42:30.618903Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.18836","last_updated":"2025-08-05T17:22:49Z","snapshot_observed_at":"2026-08-16T19:40:59.366810Z","submitted_at":"2025-02-26T05:24:22Z","title":"REALM-Bench: A Benchmark for Evaluating Multi-Agent Systems on Real-world, Dynamic Planning and Scheduling Tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.18836","snapshot_observed_at":"2026-08-16T11:42:30.623040Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.623040Z"},"links":{"cited_paper":"/paper/2502.18836","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:88c29979b16824ea6ed61809cd00f11a6216272d1c01fafcea2c2996e496252b","observation_id":"d1faf3ff-6695-493b-8e3e-5f3445e96331","resolution":{"observed_at":"2026-08-16T11:42:30.623040Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.05227","last_updated":"2025-02-06T05:50:37Z","snapshot_observed_at":"2026-08-15T19:27:00.659169Z","submitted_at":"2025-02-06T05:50:37Z","title":"Robotouille: An Asynchronous Planning Benchmark for LLM Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.05227","snapshot_observed_at":"2026-08-16T11:42:30.627423Z","title":"Robotouille: An asynchronous planning benchmark for llm agents, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.627423Z"},"links":{"cited_paper":"/paper/2502.05227","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:c7c95c3f0eee18786d2b44050175304000ce448d949816ef180b049a8250e81c","observation_id":"7759bf5f-4757-4694-a544-cc100b26917a","resolution":{"observed_at":"2026-08-16T11:42:30.627423Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1016/0004-3702(92)90028-v","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:42:30.944345Z","title":null,"venue":null,"work_id":"708b194c-4f64-4cf6-80ba-aaa38e3e76d7","year":1992},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.631643Z"},"links":{"citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:b1eda87ae37f0615940c719ca463cc482ab6ffca5327bcd3f3f142271f309345","observation_id":"ed876cd4-b325-4b55-bc97-896ac88ed323","resolution":{"observed_at":"2026-08-16T11:42:30.948573Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.06780","last_updated":"2022-02-12T20:02:13Z","snapshot_observed_at":"2026-08-16T17:56:40.043488Z","submitted_at":"2021-09-14T15:49:31Z","title":"Benchmarking the Spectrum of Agent Capabilities","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.06780","snapshot_observed_at":"2026-08-16T11:42:30.635685Z","title":"Benchmarking the spectrum of agent capabilities, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.635685Z"},"links":{"cited_paper":"/paper/2109.06780","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:8ceff075e71cf699598cfbbbf6217cd0c9e1b6af5f9c706ee0d4a50505778fc9","observation_id":"2e143f0e-a9fb-4317-a8c2-e54b4661ceb3","resolution":{"observed_at":"2026-08-16T11:42:30.635685Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.14992","last_updated":"2023-10-23T07:24:28Z","snapshot_observed_at":"2026-08-15T11:39:07.998070Z","submitted_at":"2023-05-24T10:28:28Z","title":"Reasoning with Language Model is Planning with World Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.14992","snapshot_observed_at":"2026-08-16T11:42:30.639672Z","title":"Reasoning with language model is planning with world model, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.639672Z"},"links":{"cited_paper":"/paper/2305.14992","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:a8093bed7550647223cebdf914c7ff71a2b2fa7506769bbe81c80b498b2d0eb8","observation_id":"9dbba190-bc17-427e-b823-07caf2aa6e31","resolution":{"observed_at":"2026-08-16T11:42:30.639672Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.12112","last_updated":"2025-07-09T16:13:20Z","snapshot_observed_at":"2026-08-16T19:41:00.933059Z","submitted_at":"2024-10-15T23:20:54Z","title":"Planning Anything with Rigor: General-Purpose Zero-Shot Planning with LLM-based Formalized Programming","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.12112","snapshot_observed_at":"2026-08-16T11:42:30.643822Z","title":"Planning anything with rigor: General-purpose zero-shot planning with llm-based formalized programming, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.643822Z"},"links":{"cited_paper":"/paper/2410.12112","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:724b9ec923e0c9d8846a6569b72411b51fb6c150589692942e8e906a892d732c","observation_id":"afcca4b6-03c2-4bfc-be1f-dea03a5b5865","resolution":{"observed_at":"2026-08-16T11:42:30.643822Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:42:30.648012Z","title":"Hoffmann and S","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.648012Z"},"links":{"citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:25ef459d4000786589d7dec2211bca69f142b270979885a831c329a94d718847","observation_id":"9b0de9c2-5572-4871-b4f5-5653cda51224","resolution":{"observed_at":"2026-08-16T11:42:30.648012Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.05990","last_updated":"2024-11-12T05:46:46Z","snapshot_observed_at":"2026-08-16T13:01:37.588329Z","submitted_at":"2024-11-08T22:02:22Z","title":"Game-theoretic LLM: Agent Workflow for Negotiation Games","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.05990","snapshot_observed_at":"2026-08-16T11:42:30.652939Z","title":"Game-theoretic llm: Agent workflow for negotiation games, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.652939Z"},"links":{"cited_paper":"/paper/2411.05990","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:a4b40cd424932fece5bb0a23d797d33cdc0d5ef323988bb2d3150ff6f7c094d5","observation_id":"5acfda5a-ea43-4cd1-ad6c-79bdee617df5","resolution":{"observed_at":"2026-08-16T11:42:30.652939Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.11807","last_updated":"2025-03-06T18:58:23Z","snapshot_observed_at":"2026-08-16T14:08:47.375155Z","submitted_at":"2024-03-18T14:04:47Z","title":"How Far Are We on the Decision-Making of LLMs? Evaluating LLMs' Gaming Ability in Multi-Agent Environments","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.11807","snapshot_observed_at":"2026-08-16T11:42:30.657427Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.657427Z"},"links":{"cited_paper":"/paper/2403.11807","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:b332c6ec1a6de4c26716c69c1e7403c515325a95e22e105c37e02c3720fcf11d","observation_id":"becebb25-ba08-416d-8851-3b465549bbc4","resolution":{"observed_at":"2026-08-16T11:42:30.657427Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02305","last_updated":"2025-02-16T17:16:38Z","snapshot_observed_at":"2026-08-16T19:41:19.020620Z","submitted_at":"2024-11-04T17:30:51Z","title":"CRMArena: Understanding the Capacity of LLM Agents to Perform Professional CRM Tasks in Realistic Environments","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.02305","snapshot_observed_at":"2026-08-16T11:42:30.661831Z","title":"Crmarena: Understanding the capacity of llm agents to perform professional crm tasks in realistic environments, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.661831Z"},"links":{"cited_paper":"/paper/2411.02305","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:bbf4f253164f0736e1b888e6cd915bd2ee9edd5ab5762d62dd85b520912b4419","observation_id":"5eac89ff-f8ae-49e7-8d6b-c343b7e19715","resolution":{"observed_at":"2026-08-16T11:42:30.661831Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.02716","last_updated":"2024-02-05T04:25:24Z","snapshot_observed_at":"2026-08-15T04:53:46.192738Z","submitted_at":"2024-02-05T04:25:24Z","title":"Understanding the planning of LLM agents: A survey","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.02716","snapshot_observed_at":"2026-08-16T11:42:30.665906Z","title":"Understanding the planning of llm agents: A survey, 2024 b","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.665906Z"},"links":{"cited_paper":"/paper/2402.02716","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:95d6b2ab68bb8ba982e15f2430867f70c26d428a6d04d44207e29f6c0edec239","observation_id":"9dc7b07c-eb70-4f03-a556-6dc64688e083","resolution":{"observed_at":"2026-08-16T11:42:30.665906Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:42:30.669993Z","title":"McNamara, and Deming Chen","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.669993Z"},"links":{"citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:a4b83bcee98c3e579612918bc5ed60342c35953ee30afdb58549db20a6e0d02c","observation_id":"caccdbe1-0d3e-4b59-b5fd-97c5c77d6fbd","resolution":{"observed_at":"2026-08-16T11:42:30.669993Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06770","last_updated":"2024-11-11T23:05:04Z","snapshot_observed_at":"2026-08-12T14:56:35.025839Z","submitted_at":"2023-10-10T16:47:29Z","title":"SWE-bench: Can Language Models Resolve Real-World GitHub Issues?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06770","snapshot_observed_at":"2026-08-16T11:42:30.673917Z","title":"Jimenez, John Yang, Alexander Wettig, Shunyu Yao, Kexin Pei, Ofir Press, and Karthik Narasimhan","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.673917Z"},"links":{"cited_paper":"/paper/2310.06770","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:9d0cb520d2ee5035a11545ad5de46cbe1626c4bdb910ea503f2519400820903b","observation_id":"4b81b7d5-62d8-4374-8a41-03c8649e1a1b","resolution":{"observed_at":"2026-08-16T11:42:30.673917Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.16456","last_updated":"2024-10-21T19:30:05Z","snapshot_observed_at":"2026-08-16T13:07:14.536535Z","submitted_at":"2024-10-21T19:30:05Z","title":"To the Globe (TTG): Towards Language-Driven Guaranteed Travel Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.16456","snapshot_observed_at":"2026-08-16T11:42:30.677816Z","title":"To the globe (ttg): Towards language-driven guaranteed travel planning, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.677816Z"},"links":{"cited_paper":"/paper/2410.16456","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:871c3e28d8dd25b2048a86fac959bcf751ed076c7e2463e440b343d6ebb823f8","observation_id":"9f602155-bf47-426c-8f98-f3ba1cad0d82","resolution":{"observed_at":"2026-08-16T11:42:30.677816Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:42:31.925863Z","title":"Towards a foundation for evaluating ai planners","venue":null,"work_id":"fdb08115-5261-4f6b-8a3e-c5c251c19da2","year":1990},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.681821Z"},"links":{"citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:ae3d1d79f200e258959d27aa5c2b34525606a8074bdfbf5c8186935d0d36244d","observation_id":"00bbfa85-def3-45a1-b19c-575aecf3ae68","resolution":{"observed_at":"2026-08-16T11:42:31.930089Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.13649","last_updated":"2024-06-06T02:01:09Z","snapshot_observed_at":"2026-08-15T00:56:43.641640Z","submitted_at":"2024-01-24T18:35:21Z","title":"VisualWebArena: Evaluating Multimodal Agents on Realistic Visual Web Tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.13649","snapshot_observed_at":"2026-08-16T11:42:30.686117Z","title":"Visualwebarena: Evaluating multimodal agents on realistic visual web tasks, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.686117Z"},"links":{"cited_paper":"/paper/2401.13649","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:9d88a7f37efabf37e3b3676820da36a3b2bef0a87a6505b28a142709b7de9896","observation_id":"de3cb5ef-3fa9-44f9-8d63-15a385916f7e","resolution":{"observed_at":"2026-08-16T11:42:30.686117Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14083","last_updated":"2024-04-26T21:05:19Z","snapshot_observed_at":"2026-08-17T06:38:47.622164Z","submitted_at":"2024-02-21T19:17:28Z","title":"Beyond A*: Better Planning with Transformers via Search Dynamics Bootstrapping","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.14083","snapshot_observed_at":"2026-08-16T11:42:30.690136Z","title":"Beyond a*: Better planning with transformers via search dynamics bootstrapping, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.690136Z"},"links":{"cited_paper":"/paper/2402.14083","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:5884395134bbe44f90a68e6838f8268b5d1abe7025edd04a8e01106df08bfa59","observation_id":"effb16c8-f817-4a4c-8669-f96e58f0fb2d","resolution":{"observed_at":"2026-08-16T11:42:30.690136Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01806","last_updated":"2024-09-03T11:39:52Z","snapshot_observed_at":"2026-08-16T19:41:52.976904Z","submitted_at":"2024-09-03T11:39:52Z","title":"LASP: Surveying the State-of-the-Art in Large Language Model-Assisted AI Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01806","snapshot_observed_at":"2026-08-16T11:42:30.695376Z","title":"Lasp: Surveying the state-of-the-art in large language model-assisted ai planning, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.695376Z"},"links":{"cited_paper":"/paper/2409.01806","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:3c36e2ac051acc61c386f9a354b1fc229a740e0932c2e21affd4804fce41baf3","observation_id":"8d8762b8-1d70-4513-b343-7e784854a888","resolution":{"observed_at":"2026-08-16T11:42:30.695376Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.03688","last_updated":"2025-10-04T03:54:18Z","snapshot_observed_at":"2026-08-16T00:51:32.846970Z","submitted_at":"2023-08-07T16:08:11Z","title":"AgentBench: Evaluating LLMs as Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.03688","snapshot_observed_at":"2026-08-16T11:42:30.699686Z","title":"Agentbench: Evaluating llms as agents, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.699686Z"},"links":{"cited_paper":"/paper/2308.03688","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:8584522cd6103e450e69d50578ecb8f7c65824dbf8995e49091f17dca63689a9","observation_id":"ad2e18da-06fb-4a7d-9881-341f6efbf7d9","resolution":{"observed_at":"2026-08-16T11:42:30.699686Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.13178","last_updated":"2024-12-23T20:12:48Z","snapshot_observed_at":"2026-08-16T14:25:01.697568Z","submitted_at":"2024-01-24T01:51:00Z","title":"AgentBoard: An Analytical Evaluation Board of Multi-turn LLM Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.13178","snapshot_observed_at":"2026-08-16T11:42:30.703522Z","title":"Agentboard: An analytical evaluation board of multi-turn llm agents, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.703522Z"},"links":{"cited_paper":"/paper/2401.13178","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:dc568bd2862e93b5b8a20dd5af2be37b128e1b62e2f887d2704269a7f1601d83","observation_id":"2342f4ae-9a70-4c27-aa32-a65f467b1c02","resolution":{"observed_at":"2026-08-16T11:42:30.703522Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.12983","last_updated":"2023-11-21T20:34:47Z","snapshot_observed_at":"2026-08-13T10:06:26.439949Z","submitted_at":"2023-11-21T20:34:47Z","title":"GAIA: a benchmark for General AI Assistants","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.12983","snapshot_observed_at":"2026-08-16T11:42:30.707355Z","title":"Gaia: a benchmark for general ai assistants, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.707355Z"},"links":{"cited_paper":"/paper/2311.12983","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:7c49573ed66cd6a6007d0cc8c7710bac9b7fa5358e42fba4738f9a445714ec6a","observation_id":"bd41abc8-7cfd-40e0-9b37-0378cec5168e","resolution":{"observed_at":"2026-08-16T11:42:30.707355Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10031","last_updated":"2025-01-13T06:03:14Z","snapshot_observed_at":"2026-08-16T19:41:54.250711Z","submitted_at":"2024-07-14T00:12:44Z","title":"LLaMAR: Long-Horizon Planning for Multi-Agent Robots in Partially Observable Environments","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.10031","snapshot_observed_at":"2026-08-16T11:42:30.712090Z","title":"Llamar: Long-horizon planning for multi-agent robots in partially observable environments, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.712090Z"},"links":{"cited_paper":"/paper/2407.10031","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:0c8ad56fed1d300455ff29eae6bcd6ba2c1dab082b4fafb8c5e3aedf32e14bff","observation_id":"ef09f6f1-af60-48e0-b07e-59c5772b4654","resolution":{"observed_at":"2026-08-16T11:42:30.712090Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.07778","last_updated":"2025-06-13T23:07:51Z","snapshot_observed_at":"2026-08-16T13:35:22.880915Z","submitted_at":"2024-07-10T15:52:44Z","title":"WorldAPIs: The World Is Worth How Many APIs? A Thought Experiment","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.07778","snapshot_observed_at":"2026-08-16T11:42:30.716603Z","title":"Worldapis: The world is worth how many apis? a thought experiment, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.716603Z"},"links":{"cited_paper":"/paper/2407.07778","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:82d4c7024b98705fcf00bb3ba41353e626e201e21b3847f80145bad031171279","observation_id":"27054b1d-5257-4aca-b532-c1dd74c79b9d","resolution":{"observed_at":"2026-08-16T11:42:30.716603Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.00534","last_updated":"2021-12-29T02:25:05Z","snapshot_observed_at":"2026-08-16T19:40:58.027024Z","submitted_at":"2021-10-01T17:00:14Z","title":"TEACh: Task-driven Embodied Agents that Chat","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.00534","snapshot_observed_at":"2026-08-16T11:42:30.720846Z","title":"Teach: Task-driven embodied agents that chat, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.720846Z"},"links":{"cited_paper":"/paper/2110.00534","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:c709d2dc70575d3f0ba8d55d60a9cfae30ae40f6a4206af6dd6876bb26e87ced","observation_id":"02e6f43b-b23f-4025-b65a-6f953d5a0063","resolution":{"observed_at":"2026-08-16T11:42:30.720846Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1806.07011","last_updated":"2018-06-19T02:16:44Z","snapshot_observed_at":"2026-08-15T04:16:07.694047Z","submitted_at":"2018-06-19T02:16:44Z","title":"VirtualHome: Simulating Household Activities via Programs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1806.07011","snapshot_observed_at":"2026-08-16T11:42:30.725194Z","title":"Virtualhome: Simulating household activities via programs, 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.725194Z"},"links":{"cited_paper":"/paper/1806.07011","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:0264d2975cb2a34da26f77547e5a3098dc209e689e404123805060656590ccf8","observation_id":"1270d246-03f5-49a0-971f-c7c87a248c62","resolution":{"observed_at":"2026-08-16T11:42:30.725194Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:42:31.912494Z","title":"Artificial I ntelligence: A modern approach","venue":null,"work_id":"33466c97-6600-45da-8da4-421edb61be0b","year":1995},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.729516Z"},"links":{"citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:fe22bab78bfe97fdab6b46af7c6c235dd3b59c1acbbd35967991c901362afbd5","observation_id":"4df2c259-59b7-45bc-8d13-36a89e34f829","resolution":{"observed_at":"2026-08-16T11:42:31.916816Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:42:30.733203Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.733203Z"},"links":{"citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:f50f6faaf76295e80518799057e1629d7741976cb4d8813be56ed03c65a41e05","observation_id":"bcccf922-b9f6-4e6f-85b1-70b95ae322d1","resolution":{"observed_at":"2026-08-16T11:42:30.733203Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.01240","last_updated":"2023-03-02T03:54:28Z","snapshot_observed_at":"2026-08-16T16:27:11.895195Z","submitted_at":"2022-10-03T21:34:32Z","title":"Language Models Are Greedy Reasoners: A Systematic Formal Analysis of Chain-of-Thought","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.01240","snapshot_observed_at":"2026-08-16T11:42:30.737180Z","title":"Language models are greedy reasoners: A systematic formal analysis of chain-of-thought, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.737180Z"},"links":{"cited_paper":"/paper/2210.01240","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:6d4a7fbf6c89f2c8441c2c7c944fed9a035f72e02c270d63d9321a2318a0489b","observation_id":"fe945321-7320-42a0-9d3b-304a3af33e46","resolution":{"observed_at":"2026-08-16T11:42:30.737180Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:42:30.741392Z","title":"Reflexion: Language agents with verbal reinforcement learning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.741392Z"},"links":{"citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:b628fb724d5a89c4b8e39358ae5278659538d75b05b0f8b35a1c213106a27fca","observation_id":"0ac9f86e-7e86-43ca-871f-adcbb3cef6f7","resolution":{"observed_at":"2026-08-16T11:42:30.741392Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1912.01734","last_updated":"2020-03-31T01:18:33Z","snapshot_observed_at":"2026-08-16T19:41:20.153905Z","submitted_at":"2019-12-03T23:18:59Z","title":"ALFRED: A Benchmark for Interpreting Grounded Instructions for Everyday Tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1912.01734","snapshot_observed_at":"2026-08-16T11:42:30.745222Z","title":"Alfred: A benchmark for interpreting grounded instructions for everyday tasks, 2020 a","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.745222Z"},"links":{"cited_paper":"/paper/1912.01734","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:a86bb4393c3a441b1034be14c30047a88a928a680da6f9e2602a261dc5d35d04","observation_id":"1cf8db43-aeec-4025-be1f-3cd0991f6f64","resolution":{"observed_at":"2026-08-16T11:42:30.745222Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.03768","last_updated":"2021-03-14T22:44:38Z","snapshot_observed_at":"2026-08-13T15:27:25.430393Z","submitted_at":"2020-10-08T05:13:36Z","title":"ALFWorld: Aligning Text and Embodied Environments for Interactive Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.03768","snapshot_observed_at":"2026-08-16T11:42:30.753570Z","title":"Alfworld: Aligning text and embodied environments for interactive learning, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.753570Z"},"links":{"cited_paper":"/paper/2010.03768","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:13ef4b62fc77b29ac54ab70fcbb0c512931e35f260f49bd109800e300f4670ff","observation_id":"f079fb56-3404-4631-b8de-a847da033a08","resolution":{"observed_at":"2026-08-16T11:42:30.753570Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.09918","last_updated":"2025-07-11T03:52:42Z","snapshot_observed_at":"2026-08-16T13:09:53.102837Z","submitted_at":"2024-10-13T16:53:02Z","title":"Dualformer: Controllable Fast and Slow Thinking by Learning with Randomized Reasoning Traces","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.09918","snapshot_observed_at":"2026-08-16T11:42:30.757325Z","title":"Dualformer: Controllable fast and slow thinking by learning with randomized reasoning traces, 2024 a","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.757325Z"},"links":{"cited_paper":"/paper/2410.09918","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:6c64a7126daa0cac0371603120e686ebf1941cd3521c8572c53049aeb2611042","observation_id":"eb34491c-4abb-4cea-8aa6-c7fc90b09bdf","resolution":{"observed_at":"2026-08-16T11:42:30.757325Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2024.emnlp-main.833","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:42:30.912633Z","title":"A ct P lan-1 K : Benchmarking the procedural planning ability of visual language models in household activities","venue":null,"work_id":"75846c71-f322-41c8-9343-1d308fa27f65","year":2024},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.761635Z"},"links":{"citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:64492be7884e78d29355dcbb2cddbaa082831e9807d5546717ed5b492560d99d","observation_id":"22f22cfb-abaf-4fd9-bac9-82bbb914c959","resolution":{"observed_at":"2026-08-16T11:42:30.918517Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.16419","last_updated":"2025-08-21T19:14:40Z","snapshot_observed_at":"2026-08-11T13:10:23.709172Z","submitted_at":"2025-03-20T17:59:38Z","title":"Stop Overthinking: A Survey on Efficient Reasoning for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.16419","snapshot_observed_at":"2026-08-16T11:42:30.765562Z","title":"Stop overthinking: A survey on efficient reasoning for large language models, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.765562Z"},"links":{"cited_paper":"/paper/2503.16419","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:3cd211c0db84545bde4230b8912f75636f0d00fdc4b9c5e69152e7b96ab325d4","observation_id":"af7c1c86-79ce-42d9-b11a-a338691e9d9e","resolution":{"observed_at":"2026-08-16T11:42:30.765562Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:42:31.891202Z","title":"Planbench: An extensible benchmark for evaluating large language models on planning and reasoning about change","venue":null,"work_id":"ce1b86c1-4c4d-4b76-a13a-41452d2035d3","year":2023},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.770331Z"},"links":{"citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:9954f32a35add9deb7501d503ba7855f5972a5aa4ee974afe98c7d0c416df187","observation_id":"b289fbf6-e96b-4180-87ea-a1f0e017baae","resolution":{"observed_at":"2026-08-16T11:42:31.895495Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.10479","last_updated":"2025-05-27T14:29:54Z","snapshot_observed_at":"2026-08-16T19:42:22.781139Z","submitted_at":"2024-10-14T13:15:34Z","title":"TMGBench: A Systematic Game Benchmark for Evaluating Strategic Reasoning Abilities of LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.10479","snapshot_observed_at":"2026-08-16T11:42:30.774151Z","title":"Tmgbench: A systematic game benchmark for evaluating strategic reasoning abilities of llms, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.774151Z"},"links":{"cited_paper":"/paper/2410.10479","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:80b76f52b5116706bf3d97874f5bdd3d19d45c0a2a9c26d397b69161de30642a","observation_id":"e19cb867-08ee-43b7-82b8-c8e95f7d22aa","resolution":{"observed_at":"2026-08-16T11:42:30.774151Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.14879","last_updated":"2023-10-23T18:10:31Z","snapshot_observed_at":"2026-08-16T19:42:26.682301Z","submitted_at":"2023-05-24T08:31:30Z","title":"ByteSized32: A Corpus and Challenge Task for Generating Task-Specific World Models Expressed as Text Games","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.14879","snapshot_observed_at":"2026-08-16T11:42:30.778162Z","title":"Bytesized32: A corpus and challenge task for generating task-specific world models expressed as text games, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.778162Z"},"links":{"cited_paper":"/paper/2305.14879","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:31d38ab553c26d733265ecaff67f96e735e17153e1e0d6570cc7d3c1d022a0af","observation_id":"610b3a32-17dd-4587-ba1b-dca90f4805c7","resolution":{"observed_at":"2026-08-16T11:42:30.778162Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.11221","last_updated":"2025-06-23T05:32:12Z","snapshot_observed_at":"2026-08-16T19:41:00.423849Z","submitted_at":"2025-02-16T17:54:57Z","title":"PlanGenLLMs: A Modern Survey of LLM Planning Capabilities","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.11221","snapshot_observed_at":"2026-08-16T11:42:30.782744Z","title":"PlanGenLLMs : A modern survey of llm planning capabilities, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.782744Z"},"links":{"cited_paper":"/paper/2502.11221","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:ec3180db86686a77715d499f95c3fd5c3361c3121cba588f72ba32ad932ee305","observation_id":"a41e4c9f-9017-4045-b2a3-23ae75f6fa59","resolution":{"observed_at":"2026-08-16T11:42:30.782744Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01557","last_updated":"2024-03-17T23:23:31Z","snapshot_observed_at":"2026-08-16T19:41:52.614571Z","submitted_at":"2023-10-02T18:52:11Z","title":"SmartPlay: A Benchmark for LLMs as Intelligent Agents","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01557","snapshot_observed_at":"2026-08-16T11:42:30.786968Z","title":"Mitchell, and Yuanzhi Li","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.786968Z"},"links":{"cited_paper":"/paper/2310.01557","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:8aecd2b1d463010aec83b9d0d342c32db719cd01a3896280a5a3c7bb1084b87f","observation_id":"8e5c67e6-ed17-4d1c-be77-e217a9e33843","resolution":{"observed_at":"2026-08-16T11:42:30.786968Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.02238","last_updated":"2025-03-04T03:27:02Z","snapshot_observed_at":"2026-08-16T19:41:57.237813Z","submitted_at":"2025-03-04T03:27:02Z","title":"Haste Makes Waste: Evaluating Planning Abilities of LLMs for Efficient and Feasible Multitasking with Time Constraints Between Actions","version":1},"cited_work":{"arxiv_id":"2503.02238","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.02238","snapshot_observed_at":"2026-08-16T11:42:31.213438Z","title":"Haste Makes Waste: Evaluating Planning Abilities of LLMs for Efficient and Feasible Multitasking with Time Constraints Between Actions","venue":"cs.CL","work_id":"a7c5024d-5f53-431d-a029-40f2ad7760bf","year":2025},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.791026Z"},"links":{"cited_paper":"/paper/2503.02238","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:a1c74cfca18f3e971d8a3800ec43375c193c1fcebba4f198e4eb15684841889d","observation_id":"2d1a69f6-afe7-4a6a-b2a9-260deb31ece3","resolution":{"observed_at":"2026-08-16T11:42:31.218150Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04151","last_updated":"2024-06-06T15:15:41Z","snapshot_observed_at":"2026-08-16T13:45:29.257000Z","submitted_at":"2024-06-06T15:15:41Z","title":"AgentGym: Evolving Large Language Model-based Agents across Diverse Environments","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04151","snapshot_observed_at":"2026-08-16T11:42:30.795512Z","title":"Agentgym: Evolving large language model-based agents across diverse environments, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.795512Z"},"links":{"cited_paper":"/paper/2406.04151","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:68a1867820765ea57398b2bec16cc5852352ce5f28933c5687842d1ab7b85f3d","observation_id":"877a497b-2c2c-4508-900b-19af58227af9","resolution":{"observed_at":"2026-08-16T11:42:30.795512Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01622","last_updated":"2024-10-23T15:02:57Z","snapshot_observed_at":"2026-08-16T14:22:02.195673Z","submitted_at":"2024-02-02T18:39:51Z","title":"TravelPlanner: A Benchmark for Real-World Planning with Language Agents","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01622","snapshot_observed_at":"2026-08-16T11:42:30.799753Z","title":"Travelplanner: A benchmark for real-world planning with language agents, 2024 a","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.799753Z"},"links":{"cited_paper":"/paper/2402.01622","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:b79fcf25df66076e66b754c163ae70a5903d1ed8f3bb355e85d2011b8a87e769","observation_id":"613c7175-759a-412e-aa26-ba6f541983ef","resolution":{"observed_at":"2026-08-16T11:42:30.799753Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.07972","last_updated":"2024-05-30T08:55:12Z","snapshot_observed_at":"2026-08-14T22:26:00.902198Z","submitted_at":"2024-04-11T17:56:05Z","title":"OSWorld: Benchmarking Multimodal Agents for Open-Ended Tasks in Real Computer Environments","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.07972","snapshot_observed_at":"2026-08-16T11:42:30.803851Z","title":"Osworld: Benchmarking multimodal agents for open-ended tasks in real computer environments, 2024 b","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.803851Z"},"links":{"cited_paper":"/paper/2404.07972","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:32c5585aefd3622584d7b3cba6fe2a6d509992e7cd9ff24006ed2bf1cf27d67c","observation_id":"cb4d9f63-921a-487d-9260-2e52dce4171e","resolution":{"observed_at":"2026-08-16T11:42:30.803851Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.14161","last_updated":"2025-09-10T08:35:19Z","snapshot_observed_at":"2026-08-16T19:31:08.289292Z","submitted_at":"2024-12-18T18:55:40Z","title":"TheAgentCompany: Benchmarking LLM Agents on Consequential Real World Tasks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.14161","snapshot_observed_at":"2026-08-16T11:42:30.807885Z","title":"Xu, Yufan Song, Boxuan Li, Yuxuan Tang, Kritanjali Jain, Mengxue Bao, Zora Z","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.807885Z"},"links":{"cited_paper":"/paper/2412.14161","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:3c011827ee13e2af0f5af1c9a019bfc3d4ab9d4912732e34820e896b32f0d8f3","observation_id":"fd382599-bed7-4f6b-a92c-8a76ead1bad3","resolution":{"observed_at":"2026-08-16T11:42:30.807885Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19411","last_updated":"2025-02-26T18:55:42Z","snapshot_observed_at":"2026-08-16T19:40:59.138150Z","submitted_at":"2025-02-26T18:55:42Z","title":"Code to Think, Think to Code: A Survey on Code-Enhanced Reasoning and Reasoning-Driven Code Intelligence in LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.19411","snapshot_observed_at":"2026-08-16T11:42:30.812017Z","title":"Code to think, think to code: A survey on code-enhanced reasoning and reasoning-driven code intelligence in llms, 2025 a","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.812017Z"},"links":{"cited_paper":"/paper/2502.19411","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:63408602d841d299b1b785a17fcbcc8fbd5f609ce4e2676a872115a6a422c953","observation_id":"78ba1a21-c0b7-420e-bce6-d5f59afc08da","resolution":{"observed_at":"2026-08-16T11:42:30.812017Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.09560","last_updated":"2025-06-05T07:22:50Z","snapshot_observed_at":"2026-08-15T10:11:58.661690Z","submitted_at":"2025-02-13T18:11:34Z","title":"EmbodiedBench: Comprehensive Benchmarking Multi-modal Large Language Models for Vision-Driven Embodied Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.09560","snapshot_observed_at":"2026-08-16T11:42:30.816111Z","title":"Embodiedbench: Comprehensive benchmarking multi-modal large language models for vision-driven embodied agents, 2025 b","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.816111Z"},"links":{"cited_paper":"/paper/2502.09560","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:27dcc07683477f8a8f394c7e42d9a41bb1a300987fbef6e4ff5b7426fd1e7deb","observation_id":"9026ff36-f3bd-4e07-9890-70b3a0c8aa81","resolution":{"observed_at":"2026-08-16T11:42:30.816111Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2207.01206","last_updated":"2023-02-08T01:39:30Z","snapshot_observed_at":"2026-08-16T16:48:37.086822Z","submitted_at":"2022-07-04T05:30:22Z","title":"WebShop: Towards Scalable Real-World Web Interaction with Grounded Language Agents","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2207.01206","snapshot_observed_at":"2026-08-16T11:42:30.820259Z","title":"Webshop: Towards scalable real-world web interaction with grounded language agents, 2023 a","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.820259Z"},"links":{"cited_paper":"/paper/2207.01206","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:7299aaf0434d27deee9a7c966972f51387d7d2c5507b9de48165266e1f4a2ea7","observation_id":"5791e06e-b1ed-4ce3-a82e-256a228f2f7b","resolution":{"observed_at":"2026-08-16T11:42:30.820259Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.10601","last_updated":"2023-12-03T22:50:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-17T23:16:17Z","title":"Tree of Thoughts: Deliberate Problem Solving with Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.10601","snapshot_observed_at":"2026-08-16T11:42:30.824579Z","title":"Griffiths, Yuan Cao, and Karthik Narasimhan","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.824579Z"},"links":{"cited_paper":"/paper/2305.10601","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:cde19e8886d8b37abc9b2d801fdd51d6e957f6a1288136ca912f486a26363012","observation_id":"93bb9eaf-4fdd-4a47-a684-007cfe67a7ac","resolution":{"observed_at":"2026-08-16T11:42:30.824579Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:42:30.828802Z","title":"Safeagentbench: A benchmark for safe task planning of embodied llm agents, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.828802Z"},"links":{"citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:0c75fe4bb8bc6625bbe89d6ec0dfaaf56404cbc872ab21c24f943fb2adb8cc59","observation_id":"39454e2b-b2b6-4842-b595-c65d6329bd54","resolution":{"observed_at":"2026-08-16T11:42:30.828802Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15711","last_updated":"2024-10-21T15:45:31Z","snapshot_observed_at":"2026-08-16T13:32:06.421640Z","submitted_at":"2024-07-22T15:18:45Z","title":"AssistantBench: Can Web Agents Solve Realistic and Time-Consuming Tasks?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15711","snapshot_observed_at":"2026-08-16T11:42:30.832699Z","title":"Assistantbench: Can web agents solve realistic and time-consuming tasks?, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.832699Z"},"links":{"cited_paper":"/paper/2407.15711","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:11646fa449455d6815a7c9a58346058a35ded09c6abfb7581b04cb6c7faf707e","observation_id":"8f6941aa-c503-45bf-a047-3b1ca55e54de","resolution":{"observed_at":"2026-08-16T11:42:30.832699Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.15299","last_updated":"2023-08-29T13:36:45Z","snapshot_observed_at":"2026-08-16T19:41:54.567218Z","submitted_at":"2023-08-29T13:36:45Z","title":"TaskLAMA: Probing the Complex Task Understanding of Language Models","version":1},"cited_work":{"arxiv_id":"2308.15299","doi":null,"metadata_source":"pith","pith_arxiv_id":"2308.15299","snapshot_observed_at":"2026-08-16T11:42:31.016663Z","title":"TaskLAMA: Probing the Complex Task Understanding of Language Models","venue":"cs.CL","work_id":"8a88c794-d934-4c0d-be8d-72c6667a8022","year":2023},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.836841Z"},"links":{"cited_paper":"/paper/2308.15299","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:916ec9f10c38c480d0ffe00d5a60c9b5757abfff6bbe02cca8af7d3e2bbd97d5","observation_id":"d2cc9214-7e4e-45a2-af0e-8a05d3bb772c","resolution":{"observed_at":"2026-08-16T11:42:31.021415Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05252","last_updated":"2023-05-26T06:17:17Z","snapshot_observed_at":"2026-08-16T19:41:01.444723Z","submitted_at":"2023-05-09T08:19:32Z","title":"Distilling Script Knowledge from Large Language Models for Constrained Language Planning","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.05252","snapshot_observed_at":"2026-08-16T11:42:30.840903Z","title":"Distilling script knowledge from large language models for constrained language planning, 2023 b","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.840903Z"},"links":{"cited_paper":"/paper/2305.05252","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:c13166e04d4fc48978b49ff7dada522aaa7ce45f2d9d4fcdb5653b356c7fedd1","observation_id":"f4f9304a-ad9b-4ed1-9354-fa45d1a1189a","resolution":{"observed_at":"2026-08-16T11:42:30.840903Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:42:31.878275Z","title":"Learning to decompose and organize complex tasks","venue":null,"work_id":"5ea2cc30-0b71-4c37-b8cb-9ba61783c384","year":2021},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.845008Z"},"links":{"citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:8ed98f9c466ddea263e446cbc81505b052d8b94c1041a9658a7b35d551a9c3f9","observation_id":"25ca4793-3531-4e10-9260-ff078c74fba3","resolution":{"observed_at":"2026-08-16T11:42:31.882453Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:42:30.849101Z","title":"T ime A rena: Shaping efficient multitasking language agents in a time-aware simulation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.849101Z"},"links":{"citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:8aaf65c48603e8a37c2e38d85c3c0bd7b512cdf4ff02552f34a048981f1b712a","observation_id":"4f66b0b4-75bf-49d0-832a-49936e34aa8a","resolution":{"observed_at":"2026-08-16T11:42:30.849101Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04520","last_updated":"2024-06-06T21:27:35Z","snapshot_observed_at":"2026-08-16T13:45:21.167651Z","submitted_at":"2024-06-06T21:27:35Z","title":"NATURAL PLAN: Benchmarking LLMs on Natural Language Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04520","snapshot_observed_at":"2026-08-16T11:42:30.853368Z","title":"Le, Ed H","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.853368Z"},"links":{"cited_paper":"/paper/2406.04520","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:cf3098af8f81a5440a7cd1578ab8fb3a0d917c2de45742f0def358d36a525f08","observation_id":"90c3fc94-0feb-4a78-8540-f09d38be8100","resolution":{"observed_at":"2026-08-16T11:42:30.853368Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.04406","last_updated":"2024-06-06T02:51:17Z","snapshot_observed_at":"2026-08-13T11:34:11.989911Z","submitted_at":"2023-10-06T17:55:11Z","title":"Language Agent Tree Search Unifies Reasoning Acting and Planning in Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.04406","snapshot_observed_at":"2026-08-16T11:42:30.857299Z","title":"Language agent tree search unifies reasoning acting and planning in language models, 2024 a","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.857299Z"},"links":{"cited_paper":"/paper/2310.04406","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:90fd07600c5c7be82bf1204a46f22c1421e542dd62f81770408e4a1f8c026297","observation_id":"9fc1a769-5486-4c94-ab34-06c58aa9bd11","resolution":{"observed_at":"2026-08-16T11:42:30.857299Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.13854","last_updated":"2024-04-16T15:13:18Z","snapshot_observed_at":"2026-08-14T11:14:55.351653Z","submitted_at":"2023-07-25T22:59:32Z","title":"WebArena: A Realistic Web Environment for Building Autonomous Agents","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.13854","snapshot_observed_at":"2026-08-16T11:42:30.861679Z","title":"Xu, Hao Zhu, Xuhui Zhou, Robert Lo, Abishek Sridhar, Xianyi Cheng, Tianyue Ou, Yonatan Bisk, Daniel Fried, Uri Alon, and Graham Neubig","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.861679Z"},"links":{"cited_paper":"/paper/2307.13854","citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:a6e155354579b978d567074db5e569999de6f1d9fd3dbf53ef51995fec426276","observation_id":"269952f8-09ba-4982-b8e6-7cdcbd61e686","resolution":{"observed_at":"2026-08-16T11:42:30.861679Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:42:30.865939Z","title":"@esa (Ref","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.865939Z"},"links":{"citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:ca9f99d5539b94c7a37949557246abb9039cf8364536991998e30b8d5c7e4b43","observation_id":"68d164e5-a1e6-407a-8c32-9b1b54c4621e","resolution":{"observed_at":"2026-08-16T11:42:30.865939Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:42:30.870250Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.870250Z"},"links":{"citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:2a3db59a193e62cf31770d8bf4ff09ab0e3e967f46592da17aeb0649b866a158","observation_id":"cb914a45-a361-4faf-b6cc-62f0213c3d43","resolution":{"observed_at":"2026-08-16T11:42:30.870250Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T11:42:30.874258Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities","version":1},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-08-16T11:42:30.874258Z"},"links":{"citing_paper":"/paper/2504.14773"},"observation_digest":"sha256:398def51ca1c7e112c8297e8f3db08f9889e8f8034a97621795bcb23d4e230e4","observation_id":"a6de5176-4b8d-417c-8ef4-607aa6fe8af9","resolution":{"observed_at":"2026-08-16T11:42:30.874258Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2504.14773","last_updated":"2025-04-21T00:02:50Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-16T19:40:23.647414Z","submitted_at":"2025-04-21T00:02:50Z","title":"PLANET: A Collection of Benchmarks for Evaluating LLMs' Planning Capabilities"},"reference_resolution":{"displayed":73,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":64,"verified_exact":5,"verified_fuzzy":4},"total_outbound_references":73},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"thesis":"As of 17 August 2026, this Paper Citation Record lists 73 of 73 outbound references and 1 inbound Pith citation observation for arXiv:2504.14773."}