{"as_of":"2026-08-14T07:24:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:6664a33f0b5c118737d534b83ed8bc23c7ec2d133b4460d7eaa267107386ec16","coverage":[{"denominator":115,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:27:03.187728Z","state":"measured"},{"denominator":122,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":122,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-14T06:32:32.682623+00:00","state":"measured"},{"denominator":22,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":22,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T22:50:33.015304Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-07T22:50:33.015304Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2502.09053","last_updated":"2025-08-05T02:23:31Z","snapshot_observed_at":"2026-08-08T10:12:47.586452Z","submitted_at":"2025-02-13T08:08:27Z","title":"Game Theory Meets Large Language Models: A Systematic Survey with Taxonomy and New Frontiers","version":2},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-07T22:50:33.015304Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2502.09053"},"observation_digest":"sha256:6cd394cbe2aeb3bda09efb63b957e357a738994f896fbb72050c7f580d67eb26","observation_id":"6eff8939-1f26-465a-9aab-d505c04f2523","resolution":{"observed_at":"2026-08-07T22:50:33.015304Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-06T04:33:42.500139Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.03368","last_updated":"2025-08-18T09:53:16Z","snapshot_observed_at":"2026-08-07T21:23:20.919037Z","submitted_at":"2025-08-05T12:15:59Z","title":"Game Reasoning Arena: A Framework and Benchmark for Assessing Reasoning Capabilities of Large Language Models via Game Play","version":3},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-06T04:33:42.500139Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2508.03368"},"observation_digest":"sha256:3482d7075efac449a77e6076e4a672fbda7cfc4cafc53fde44737c6e64bb9a0a","observation_id":"6b6def62-bd00-4966-905d-2beeba9e4b8f","resolution":{"observed_at":"2026-08-06T04:33:42.500139Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2509.02544","last_updated":"2025-09-05T14:59:27Z","snapshot_observed_at":"2026-08-11T12:35:47.937091Z","submitted_at":"2025-09-02T17:44:45Z","title":"UI-TARS-2 Technical Report: Advancing GUI Agent with Multi-Turn Reinforcement Learning","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-13T10:13:58.774968Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2509.02544"},"observation_digest":"sha256:3f736007888f45964df9f804740e7c27ad0427d3286557bcb9d2fb5169c44b59","observation_id":"c87a4664-3ed6-421e-bf53-762d56b3599c","resolution":{"observed_at":"2026-05-13T10:13:59.184183Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2509.08827","last_updated":"2025-10-09T17:08:52Z","snapshot_observed_at":"2026-08-06T15:38:05.011922Z","submitted_at":"2025-09-10T17:59:43Z","title":"A Survey of Reinforcement Learning for Large Reasoning Models","version":3},"reference_index":203,"source":"arxiv_source","source_observed_at":"2026-05-18T00:02:24.352947Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2509.08827"},"observation_digest":"sha256:3278d3d4a3b796becdd086d5211a0ee082f92d651f6803a4c6a8035b467463ab","observation_id":"27ec31b8-d11b-4952-889c-a66ceb5987ad","resolution":{"observed_at":"2026-05-18T00:05:31.526497Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2603.15432","last_updated":"2026-04-08T15:52:59Z","snapshot_observed_at":"2026-08-13T13:13:25.945624Z","submitted_at":"2026-03-16T15:37:07Z","title":"Gym-V: A Unified Vision Environment System for Agentic Vision Research","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-15T10:05:09.049846Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2603.15432"},"observation_digest":"sha256:8c82a941328712a525c849cd620c53a0e8f3a137dfa364d900a9be174939e3f2","observation_id":"f8567ab9-4edc-49b5-a74d-036f1e741261","resolution":{"observed_at":"2026-05-15T10:05:26.044132Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2604.20043","last_updated":"2026-04-21T22:55:57Z","snapshot_observed_at":"2026-08-10T21:52:27.129411Z","submitted_at":"2026-04-21T22:55:57Z","title":"TriEx: A Game-based Tri-View Framework for Explaining Internal Reasoning in Multi-Agent LLMs","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T02:04:57.744754Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2604.20043"},"observation_digest":"sha256:e2b4dd9a101b76eb0de3a50aa4da9baa0731057260b191a523d60c736faa9d25","observation_id":"5f80c1c4-4c08-452c-a764-51911f301c29","resolution":{"observed_at":"2026-05-11T13:16:06.152542Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2604.20987","last_updated":"2026-04-22T18:17:17Z","snapshot_observed_at":"2026-08-11T10:29:19.336072Z","submitted_at":"2026-04-22T18:17:17Z","title":"Co-Evolving LLM Decision and Skill Bank Agents for Long-Horizon Tasks","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-10T00:14:07.017420Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2604.20987"},"observation_digest":"sha256:f13b50e8400a96538eee3634b68b50932304988f8d025b9c883da46b36cbeb15","observation_id":"e18a66e5-bba2-4612-b96a-11391a6046d7","resolution":{"observed_at":"2026-05-10T00:14:46.428384Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2605.00347","last_updated":"2026-05-01T02:05:56Z","snapshot_observed_at":"2026-08-11T00:38:42.624588Z","submitted_at":"2026-05-01T02:05:56Z","title":"Odysseus: Scaling VLMs to 100+ Turn Decision-Making in Games via Reinforcement Learning","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-09T20:22:58.061772Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2605.00347"},"observation_digest":"sha256:8a29c9b72db1ba25164225c91c781a4cb04589718e7b84347b7ba7433ad1fc73","observation_id":"1fb4d952-c576-432d-96bc-4165ca85794e","resolution":{"observed_at":"2026-05-11T15:16:09.364638Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2605.13527","last_updated":"2026-06-01T11:38:10Z","snapshot_observed_at":"2026-07-06T23:25:07.022065Z","submitted_at":"2026-05-13T13:40:31Z","title":"MMSkills: Towards Multimodal Skills for General Visual Agents","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-14T19:05:36.511150Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2605.13527"},"observation_digest":"sha256:2a5b2b796dd349fd87317e351ed1209049bcbb91d18055526e19567d6be90c75","observation_id":"70f766f4-694a-44b7-b27c-a41a16861d3c","resolution":{"observed_at":"2026-05-14T19:07:51.419743Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2605.13527","last_updated":"2026-06-01T11:38:10Z","snapshot_observed_at":"2026-07-06T23:25:07.022065Z","submitted_at":"2026-05-13T13:40:31Z","title":"MMSkills: Towards Multimodal Skills for General Visual Agents","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-15T05:59:44.669877Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2605.13527"},"observation_digest":"sha256:192dd995d7ffaa31dad34f5751959a76d7e7833778a66e413bcecaeaf22c6b81","observation_id":"5d47647a-68a4-4f89-984d-0df532ab97c7","resolution":{"observed_at":"2026-05-15T05:59:48.141281Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2605.13527","last_updated":"2026-06-01T11:38:10Z","snapshot_observed_at":"2026-07-06T23:25:07.022065Z","submitted_at":"2026-05-13T13:40:31Z","title":"MMSkills: Towards Multimodal Skills for General Visual Agents","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-30T21:31:20.079403Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2605.13527"},"observation_digest":"sha256:13677b085a5e8c586494e51ac5880ae190433f8b84eb8b4a63cbb0d91875f2d0","observation_id":"1d368f40-3266-41ea-8cdd-8b9fb6fa95df","resolution":{"observed_at":"2026-06-30T21:35:04.559898Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2605.17637","last_updated":"2026-05-21T20:02:32Z","snapshot_observed_at":"2026-08-03T12:30:05.317561Z","submitted_at":"2026-05-17T20:07:12Z","title":"WebGameBench: Requirement-to-Application Evaluation for Coding Agents via Browser-Native Games","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-20T12:24:06.062957Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2605.17637"},"observation_digest":"sha256:35675ff84163b4fcf3a0c9a703b24a75f65318896ef6662701adb491124fe4da","observation_id":"cfd7b7b2-11a5-41eb-8577-10df4fcd792b","resolution":{"observed_at":"2026-05-20T12:28:17.261880Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2605.17637","last_updated":"2026-05-21T20:02:32Z","snapshot_observed_at":"2026-08-03T12:30:05.317561Z","submitted_at":"2026-05-17T20:07:12Z","title":"WebGameBench: Requirement-to-Application Evaluation for Coding Agents via Browser-Native Games","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-25T05:45:04.573722Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2605.17637"},"observation_digest":"sha256:c43d08b870dbe99c893791f86d5465cf5f6000ab01092733cd9665c3ecc25084","observation_id":"2049d76d-8774-4384-a78f-ba5c6d631ed4","resolution":{"observed_at":"2026-05-25T05:45:23.221897Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2605.29512","last_updated":"2026-05-28T07:33:47Z","snapshot_observed_at":"2026-08-13T13:12:26.137402Z","submitted_at":"2026-05-28T07:33:47Z","title":"MINDGAMES: A Live Arena for Evaluating Social and Strategic Reasoning in Multi-Agent LLMs","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-29T07:15:27.939886Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2605.29512"},"observation_digest":"sha256:d76c63cdeea7715dd2f473409f4f525033b41050c36c175d45828858c016b0b0","observation_id":"cb5202f0-d7ec-4fae-b9ff-6d58f9651a63","resolution":{"observed_at":"2026-06-29T07:23:13.196706Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2605.29653","last_updated":"2026-05-28T09:16:22Z","snapshot_observed_at":"2026-08-12T07:45:20.901825Z","submitted_at":"2026-05-28T09:16:22Z","title":"PTCG-Bench: Can LLM Agents Master Pok\\'emon Trading Card Game?","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-29T07:21:49.763994Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2605.29653"},"observation_digest":"sha256:9fdb102cde40dd63b44b7cf4741d20fe060f0305a0f91b2f1e72d87cfb152e7b","observation_id":"23f86099-b0a4-427e-800a-d1b8d9bdeb5f","resolution":{"observed_at":"2026-06-29T07:23:12.593221Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2606.06556","last_updated":"2026-06-04T10:43:14Z","snapshot_observed_at":"2026-08-09T06:57:25.072034Z","submitted_at":"2026-06-04T10:43:14Z","title":"Robots Need More than VLA and World Models","version":1},"reference_index":152,"source":"arxiv_source","source_observed_at":"2026-06-28T01:01:33.530167Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2606.06556"},"observation_digest":"sha256:6a2435e1b060b45a27b66fa3ee54227e02d2ab6b391c77c4eb38cbf5882495c7","observation_id":"cad8132a-3e76-4034-aed2-e4abc28a1d66","resolution":{"observed_at":"2026-06-28T01:11:28.899789Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2606.09826","last_updated":"2026-06-08T17:59:43Z","snapshot_observed_at":"2026-08-02T20:00:31.322130Z","submitted_at":"2026-06-08T17:59:43Z","title":"OmniGameArena: A Unified UE5 Benchmark for VLM Game Agents with Improvement Dynamics","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-06-27T16:50:36.194650Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2606.09826"},"observation_digest":"sha256:6bb3a47d68f7d4b7fec7bffadb4e8f038dc0955294a55e3cd0580945a8ec8256","observation_id":"f4c3ebe9-e250-4a38-9fe2-063b44fa6a8c","resolution":{"observed_at":"2026-07-03T01:07:29.947294Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2606.12191","last_updated":"2026-06-10T15:15:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-06-10T15:15:01Z","title":"Agentic Environment Engineering for Large Language Models: A Survey of Environment Modeling, Synthesis, Evaluation, and Application","version":1},"reference_index":116,"source":"pdf_text","source_observed_at":"2026-06-27T09:46:30.702256Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2606.12191"},"observation_digest":"sha256:230fa132ede68ce9961ea550ffc8735c3e031bbe09f74af582eccf5b6b1e187b","observation_id":"09f90fa0-57f8-4c7f-9529-c4a8dd5f4cc1","resolution":{"observed_at":"2026-06-27T09:50:48.495612Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2606.18950","last_updated":"2026-06-18T08:25:27Z","snapshot_observed_at":"2026-08-13T18:22:55.704603Z","submitted_at":"2026-06-17T11:32:51Z","title":"RTSGameBench: An RTS Benchmark for Strategic Reasoning by Vision-Language Models","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-06-26T21:02:26.081166Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2606.18950"},"observation_digest":"sha256:4abfef8eaaeecde3d7dcf010e2138605ac75c18fbc8fc18b775788d90b5aae9b","observation_id":"0be5c38e-8540-42a0-be0c-76e3c213b0e2","resolution":{"observed_at":"2026-07-04T00:39:17.299890Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2606.19338","last_updated":"2026-06-17T17:59:34Z","snapshot_observed_at":"2026-07-06T23:54:37.596257Z","submitted_at":"2026-06-17T17:59:34Z","title":"Beyond the Current Observation: Evaluating Multimodal Large Language Models in Controllable Non-Markov Games","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-06-26T21:17:02.332687Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2606.19338"},"observation_digest":"sha256:3c998354dc03ade507c387f5d11fe2d59fe879ce3a80f3a09f0d07dadb760261","observation_id":"914f01fc-110b-4bcd-9f5a-dbb5aad4ede8","resolution":{"observed_at":"2026-07-04T00:19:13.718478Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2606.24893","last_updated":"2026-05-29T22:40:51Z","snapshot_observed_at":"2026-08-14T00:29:24.445166Z","submitted_at":"2026-05-29T22:40:51Z","title":"AgentOdyssey: Open-Ended Long-Horizon Text Game Generation for Test-Time Continual Learning Agents","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-28T21:59:25.449570Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2606.24893"},"observation_digest":"sha256:3a9f9fbf296cff685336c03dbe70718e5a72da5acd1512f0055d61360f028df8","observation_id":"0c2486de-cfd0-4b84-97a8-f3b4a71f6f00","resolution":{"observed_at":"2026-07-01T19:46:11.447161Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-01T02:58:29.283205Z","title":"arXiv preprint arXiv:2505.15146 , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.25308","last_updated":"2026-07-28T05:39:01Z","snapshot_observed_at":"2026-08-05T09:28:39.486019Z","submitted_at":"2026-07-28T05:39:01Z","title":"CAST: Game Solvers as Turn-Level Teachers for LLM Agents","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-01T02:58:29.283205Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2607.25308"},"observation_digest":"sha256:796ea5191f1050b41e2acd7a2247a7421da7cd39ecedb76a43708abff05abf55","observation_id":"c51221f7-15e7-4980-ac29-d6e0cb128700","resolution":{"observed_at":"2026-08-01T02:58:29.283205Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2505.15146/citation-record","integrity":"/paper/2505.15146/integrity","json":"/paper/2505.15146/citation-record.json","paper":"/paper/2505.15146"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"1606.01540","last_updated":"2016-06-05T17:54:48Z","snapshot_observed_at":"2026-08-13T12:26:05.192883Z","submitted_at":"2016-06-05T17:54:48Z","title":"OpenAI Gym","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1606.01540","snapshot_observed_at":"2026-08-07T15:26:55.977387Z","title":"arXiv preprint arXiv:1606.01540 (2016)","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:55.977387Z"},"links":{"cited_paper":"/paper/1606.01540","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:3205e97e07aa1f21f6baa372bee5959c66674bd538f8542fd67e319867a410bf","observation_id":"a7e70288-ebc0-4e09-82a6-f0ea6f228c64","resolution":{"observed_at":"2026-08-07T15:26:55.977387Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.17032","last_updated":"2025-11-02T13:42:19Z","snapshot_observed_at":"2026-08-13T22:24:37.672685Z","submitted_at":"2024-07-24T06:35:05Z","title":"Gymnasium: A Standard Interface for Reinforcement Learning Environments","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.17032","snapshot_observed_at":"2026-08-07T15:26:56.072057Z","title":"arXiv preprint arXiv:2407.17032 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:56.072057Z"},"links":{"cited_paper":"/paper/2407.17032","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:a0373c70804f422343bb2ddd5a8753f240a7c71c6da11a9b931396f9b42ed365","observation_id":"131c4169-85b2-400a-865b-adda2c05762f","resolution":{"observed_at":"2026-08-07T15:26:56.072057Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-08-13T08:41:26.093022Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-07T15:26:56.140833Z","title":"arXiv preprint arXiv:2504.20073 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:56.140833Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:9fb40a6ce5795c97d0d6f85222fb6152e4b533b2517763112f53a85527537765","observation_id":"ba224b0f-9985-4a21-bb0e-6fa7100a6c5c","resolution":{"observed_at":"2026-08-07T15:26:56.140833Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.15478","last_updated":"2025-03-19T17:55:08Z","snapshot_observed_at":"2026-08-08T21:39:18.091217Z","submitted_at":"2025-03-19T17:55:08Z","title":"SWEET-RL: Training Multi-Turn LLM Agents on Collaborative Reasoning Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.15478","snapshot_observed_at":"2026-08-07T15:26:56.255061Z","title":"arXiv preprint arXiv:2503.15478 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:56.255061Z"},"links":{"cited_paper":"/paper/2503.15478","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:3de208a9bb9b73809babb622041ed4e27513f586c59282604992da82d8b6e2ee","observation_id":"14a9e272-3755-4f82-8ebb-4c5d0b603de9","resolution":{"observed_at":"2026-08-07T15:26:56.255061Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:56.363888Z","title":"In Globerson, A., Mackey, L., Belgrave, D., Fan, A., Paquet, U., Tomczak, J., Zhang, C., eds.: Advances in Neural Information Processing Systems","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:56.363888Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:7290f23059c849addd6eb4cdd614925d0605b06aedd61920ae963c03cea37d4b","observation_id":"1d9d2fe6-ad2e-4a8c-ae80-807da734ae0b","resolution":{"observed_at":"2026-08-07T15:26:56.363888Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:56.453904Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:56.453904Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:1dd46f1bfa8808db32c92e60cfed3f4abc0a4749c51d6fddd335316f2eeea4f1","observation_id":"6d0d7fbc-84b1-43ad-a74d-6b37e571ae55","resolution":{"observed_at":"2026-08-07T15:26:56.453904Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:56.552570Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:56.552570Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:109eb338015f8f44a6c29b8f6a882526b1939c7b186c96d6fa0a7b099b4ce018","observation_id":"babc8d0b-adff-4d21-9ae3-ee3d6a759645","resolution":{"observed_at":"2026-08-07T15:26:56.552570Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-11T19:06:15.885413Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.08859","snapshot_observed_at":"2026-08-07T15:26:56.639836Z","title":"arXiv preprint arXiv:2502.08859 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:56.639836Z"},"links":{"cited_paper":"/paper/2502.08859","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:f23e39df500e11fd2d3b832ee15de6fc70c2b51e77ed29d01e0f6682d6422fff","observation_id":"57530e12-cda5-41fe-bf7f-6574601f213e","resolution":{"observed_at":"2026-08-07T15:26:56.639836Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:56.720112Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:56.720112Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:41601f1149d4b0470821a97f9f92f20576b8351cfa4b7400996208fafd48bffa","observation_id":"a0ca5e66-1c30-491b-9df9-19c7522ef01d","resolution":{"observed_at":"2026-08-07T15:26:56.720112Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.13543","last_updated":"2025-04-01T14:45:22Z","snapshot_observed_at":"2026-08-13T22:17:36.984752Z","submitted_at":"2024-11-20T18:54:32Z","title":"BALROG: Benchmarking Agentic LLM and VLM Reasoning On Games","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.13543","snapshot_observed_at":"2026-08-07T15:26:56.816724Z","title":"arXiv preprint arXiv:2411.13543 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:56.816724Z"},"links":{"cited_paper":"/paper/2411.13543","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:d41eb1a8e462d94b768849fc3ca8388a4ab60a6067d252b165014f8cf6b7e785","observation_id":"fc84ddec-7530-45b2-aa78-08091fdbc4a6","resolution":{"observed_at":"2026-08-07T15:26:56.816724Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.06613","last_updated":"2024-07-22T14:32:33Z","snapshot_observed_at":"2026-08-12T23:48:17.108802Z","submitted_at":"2024-06-07T00:28:43Z","title":"GameBench: Evaluating Strategic Reasoning Abilities of LLM Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.06613","snapshot_observed_at":"2026-08-07T15:26:56.894546Z","title":"arXiv preprint arXiv:2406.06613 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:56.894546Z"},"links":{"cited_paper":"/paper/2406.06613","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:7ee7458304ba7dfb5d1be472a50bbc4602ca3d599f852a1911ed00cb1cfc8e4f","observation_id":"7e38bd61-6f22-4a25-8709-7c5227b6d909","resolution":{"observed_at":"2026-08-07T15:26:56.894546Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01557","last_updated":"2024-03-17T23:23:31Z","snapshot_observed_at":"2026-08-13T05:58:59.275924Z","submitted_at":"2023-10-02T18:52:11Z","title":"SmartPlay: A Benchmark for LLMs as Intelligent Agents","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01557","snapshot_observed_at":"2026-08-07T15:26:57.020933Z","title":"arXiv preprint arXiv:2310.01557 (2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:57.020933Z"},"links":{"cited_paper":"/paper/2310.01557","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:7f5e8b1806f5260249a1c6e073491c7f8fc2f8075c1a50cdc149f81493ad32e5","observation_id":"feb50f57-f0b3-4573-880d-350a08c99f3b","resolution":{"observed_at":"2026-08-07T15:26:57.020933Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:57.127873Z","title":"IEEE Transactions on Games11(3) (2019) 195–202","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:57.127873Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:dcf09d9aa6fb3a49bb8a80596058fb43411781f85667593522e11db96678cf68","observation_id":"6aa6771e-9efd-40a8-8ccd-f8cd596c8f53","resolution":{"observed_at":"2026-08-07T15:26:57.127873Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:57.204676Z","title":"AI Magazine22(2) (2001) 15–25","venue":null,"work_id":null,"year":2001},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:57.204676Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:5d00c0e381d3674a7e2706174bf8ad170dcf8a3a4dcaae22863cc136f23abef8","observation_id":"97837f3f-1fdd-4b8d-9c47-82e019956ede","resolution":{"observed_at":"2026-08-07T15:26:57.204676Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.14171","last_updated":"2025-07-02T21:00:36Z","snapshot_observed_at":"2026-08-10T08:02:13.965614Z","submitted_at":"2024-12-18T18:59:54Z","title":"Thinking in Space: How Multimodal Large Language Models See, Remember, and Recall Spaces","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.14171","snapshot_observed_at":"2026-08-07T15:26:57.341174Z","title":"arXiv preprint arXiv:2412.14171 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:57.341174Z"},"links":{"cited_paper":"/paper/2412.14171","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:bba950304ee666d812281b18905b62f7da0a95de93e8506e8556f95c90c96d78","observation_id":"14041c05-4acb-4b6b-bb51-360566b36a2b","resolution":{"observed_at":"2026-08-07T15:26:57.341174Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15950","last_updated":"2024-12-02T03:48:43Z","snapshot_observed_at":"2026-08-12T22:55:44.535039Z","submitted_at":"2024-08-28T17:08:56Z","title":"Atari-GPT: Benchmarking Multimodal Large Language Models as Low-Level Policies in Atari Games","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15950","snapshot_observed_at":"2026-08-07T15:26:57.413647Z","title":"arXiv preprint arXiv:2408.15950 (2024) 11","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:57.413647Z"},"links":{"cited_paper":"/paper/2408.15950","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:6c6b5524892d0318770888b36a913111aa5bf95a75e61df48df9be22e4f5ef58","observation_id":"2607c706-b810-4f3e-a02e-76c5f1a9e56e","resolution":{"observed_at":"2026-08-07T15:26:57.413647Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.11381","last_updated":"2024-06-19T16:23:05Z","snapshot_observed_at":"2026-08-13T00:52:20.918062Z","submitted_at":"2024-03-18T00:13:43Z","title":"Can LLM-Augmented autonomous agents cooperate?, An evaluation of their cooperative capabilities through Melting Pot","version":2},"cited_work":{"arxiv_id":"2403.11381","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.11381","snapshot_observed_at":"2026-08-07T15:27:03.880129Z","title":"Can LLM-Augmented autonomous agents cooperate?, An evaluation of their cooperative capabilities through Melting Pot","venue":"cs.AI","work_id":"03b139c8-2038-4fb7-a36c-fca8d92ab30a","year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:57.507753Z"},"links":{"cited_paper":"/paper/2403.11381","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:33abaf8e29ce6662aee6bc1b3707d7d2a1c23391bec0cc8c063b2aa0d116106e","observation_id":"42618b84-232d-4f90-b0d7-8e954d3735c2","resolution":{"observed_at":"2026-08-07T15:27:03.886471Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:57.604725Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:57.604725Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:1912417122a40c80125fcd9f28efa87fe5b8fabec4666126357a996ebb982002","observation_id":"0e90353b-7af8-4f73-8b05-213dc2a49d13","resolution":{"observed_at":"2026-08-07T15:26:57.604725Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.16720","last_updated":"2026-04-30T02:46:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-21T18:04:31Z","title":"OpenAI o1 System Card","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.16720","snapshot_observed_at":"2026-08-07T15:26:57.709482Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:57.709482Z"},"links":{"cited_paper":"/paper/2412.16720","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:d0de09aa5affd6839e26c44e549c6681b7404a2cc1003740b25d6b02c666cb0f","observation_id":"3c06fd2b-4bce-456f-ae46-6b275c790620","resolution":{"observed_at":"2026-08-07T15:26:57.709482Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:57.792901Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:57.792901Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:09eda6388a36b92860f7b1c1db4ea06c1168122db1c896cb8ac590738779bbc3","observation_id":"e2613515-e9ef-4538-b0cb-848ce7429f6a","resolution":{"observed_at":"2026-08-07T15:26:57.792901Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:57.864760Z","title":"In ICAPS","venue":null,"work_id":null,"year":2004},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:57.864760Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:5fa127b4a5b9151dcc41cb965dc5b83be6b94fff02f5262d6e6ef2d44a7b35f5","observation_id":"f282f996-6b9a-4e1f-8578-ad4a26aa3d85","resolution":{"observed_at":"2026-08-07T15:26:57.864760Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:57.936132Z","title":"Applied cognitive psychology31(4) (2017) 438–445","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:57.936132Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:2b52d343bdcb362881c088cdf047d9e30effa5892c774cc099782cb46a9d361e","observation_id":"5ed9dfa1-d833-40fa-af0e-82ebf70154f0","resolution":{"observed_at":"2026-08-07T15:26:57.936132Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:58.046424Z","title":"In International Computing and Combinatorics Conference (COCOON)","venue":null,"work_id":null,"year":2003},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:58.046424Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:33e58bf522cf27e70f06706f5c3fee0b4f25fc0007aa419daf5ebc98daec8b27","observation_id":"a3791a52-c555-4ec5-bb2f-a3ea2101db68","resolution":{"observed_at":"2026-08-07T15:26:58.046424Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:58.163042Z","title":null,"venue":null,"work_id":null,"year":1997},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:58.163042Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:9d54e5bf35a81d516a4d4dd53d083727f6220e82731838533d4f5449d96b8092","observation_id":"a661cdd9-5b21-433f-b165-7d7818203605","resolution":{"observed_at":"2026-08-07T15:26:58.163042Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:58.220901Z","title":null,"venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:58.220901Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:ed150d70f9cd5d43c1901c9353a3daba723f5a1fc8dc76e9e596d1b29847c467","observation_id":"458995a7-b87c-48cf-892d-495c8af4cd2b","resolution":{"observed_at":"2026-08-07T15:26:58.220901Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:58.280311Z","title":"arXiv (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:58.280311Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:c38943cc6277e3ab7c04e8f3bf2ec690de6c50ddba1a5b8726f9b6661d5b91c6","observation_id":"68d50081-a23e-4b11-ad68-4a9665b646da","resolution":{"observed_at":"2026-08-07T15:26:58.280311Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:58.353668Z","title":"arXiv (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:58.353668Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:4a9116897b55ab077978ea9c6ae1a54a71a2dd5844c27992207f64f18d17537c","observation_id":"eadfffa4-9db8-4542-9f36-cc139a0416a3","resolution":{"observed_at":"2026-08-07T15:26:58.353668Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-08-13T15:58:13.809876Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-07T15:26:58.427826Z","title":"arXiv preprint arXiv:2501.12948 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:58.427826Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:24b2ac846e172ead943b2a58d56919f268e3fe10f1411b3d5f3d094dc9a497b5","observation_id":"b9973ff4-17fc-4c65-a76e-ebac676753b3","resolution":{"observed_at":"2026-08-07T15:26:58.427826Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:58.487599Z","title":"Computational Geometry 13(4) (1999) 215–228","venue":null,"work_id":null,"year":1999},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:58.487599Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:f5faed9ee41ff880e70bc4644913ad29781cd74fa1176514b4fbb9959f9c803d","observation_id":"24d32840-27ff-4a3a-95c4-4e9206bb84f4","resolution":{"observed_at":"2026-08-07T15:26:58.487599Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1403.5484","last_updated":"2014-03-21T15:03:33Z","snapshot_observed_at":"2026-07-06T03:39:07.358584Z","submitted_at":"2014-03-21T15:03:33Z","title":"A Simple Family of Analytical Trumpet Slices of the Schwarzschild Spacetime","version":1},"cited_work":{"arxiv_id":"1403.5484","doi":null,"metadata_source":"pith","pith_arxiv_id":"1403.5484","snapshot_observed_at":"2026-08-07T15:27:03.815957Z","title":"A Simple Family of Analytical Trumpet Slices of the Schwarzschild Spacetime","venue":"gr-qc","work_id":"f4d774ae-ac75-4564-b1a9-c2dddd2e2384","year":2014},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:58.543471Z"},"links":{"cited_paper":"/paper/1403.5484","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:d8c0c026e5dc59c08046879c4783bc5f70ce5cb6ed94f422aded0522d4699e28","observation_id":"315386b0-a926-4c90-b7d8-da2b71f4b8b2","resolution":{"observed_at":"2026-08-07T15:27:03.824241Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15938","last_updated":"2024-05-31T17:49:03Z","snapshot_observed_at":"2026-08-13T04:10:47.901686Z","submitted_at":"2024-02-24T23:54:41Z","title":"Generalization or Memorization: Data Contamination and Trustworthy Evaluation for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.15938","snapshot_observed_at":"2026-08-07T15:26:58.602648Z","title":"arXiv preprint arXiv:2402.15938 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:58.602648Z"},"links":{"cited_paper":"/paper/2402.15938","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:7653fa64fad39fa61e5e4190cd12697a24f9c1246e78a4a9f0a99061bf02d5c5","observation_id":"df103d95-f26f-4379-a55b-5e402af91911","resolution":{"observed_at":"2026-08-07T15:26:58.602648Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1802.08232","last_updated":"2019-07-16T17:05:32Z","snapshot_observed_at":"2026-08-06T23:45:16.145644Z","submitted_at":"2018-02-22T18:42:41Z","title":"The Secret Sharer: Evaluating and Testing Unintended Memorization in Neural Networks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1802.08232","snapshot_observed_at":"2026-08-07T15:26:58.673836Z","title":"arXiv preprint arXiv:1802.08232 (2018)","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:58.673836Z"},"links":{"cited_paper":"/paper/1802.08232","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:bcb0a0e7281f9b5ebc86e535b36d397bed7f36ad3abd6ed3bb061f20f4abb992","observation_id":"b6f11591-6ffc-4ad2-a2a3-a1758280f791","resolution":{"observed_at":"2026-08-07T15:26:58.673836Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:58.734689Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:58.734689Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:08d495f9229aa89824ebe010de421a40d5005fd7225642ced1ac64c9bf286071","observation_id":"779eca80-188b-4c50-852e-178d48195594","resolution":{"observed_at":"2026-08-07T15:26:58.734689Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:58.819840Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:58.819840Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:03b5aab34c5745fbcdb3ac15655f1eb276b0c4a3ba9f9ae9600b6ebc70a65c90","observation_id":"c90c83f6-52ab-4545-84d1-3304d47f63c3","resolution":{"observed_at":"2026-08-07T15:26:58.819840Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.03186","last_updated":"2024-07-02T17:23:13Z","snapshot_observed_at":"2026-08-13T04:02:49.444355Z","submitted_at":"2024-03-05T18:22:29Z","title":"Cradle: Empowering Foundation Agents Towards General Computer Control","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.03186","snapshot_observed_at":"2026-08-07T15:26:58.872288Z","title":"arXiv preprint arXiv:2403.03186 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:58.872288Z"},"links":{"cited_paper":"/paper/2403.03186","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:c390344a33cc72477e53f0e8230b5bb954077a5123c063d2fccfb7148fae7fe2","observation_id":"5a38610f-7acc-42ac-a56d-c9ab3c9d4908","resolution":{"observed_at":"2026-08-07T15:26:58.872288Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:58.941657Z","title":"In The Twelfth International Conference on Learning Representations","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:58.941657Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:4f0853c7708340af1d4e7469d015c26ad71617b0c8285a6e382f81df1f928e51","observation_id":"091a4179-5910-422c-8688-a32d3cb84c84","resolution":{"observed_at":"2026-08-07T15:26:58.941657Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:59.038784Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:59.038784Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:bd7c9fdb9aeab9861f38a64cb55fee8581f2536f56af010890df18417debf516","observation_id":"e14d8a20-afe8-440c-b5d0-b4b75a82f599","resolution":{"observed_at":"2026-08-07T15:26:59.038784Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2009.03300","last_updated":"2021-01-12T18:57:11Z","snapshot_observed_at":"2026-08-13T20:44:28.824685Z","submitted_at":"2020-09-07T17:59:25Z","title":"Measuring Massive Multitask Language Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.03300","snapshot_observed_at":"2026-08-07T15:26:59.115491Z","title":"arXiv preprint arXiv:2009.03300 (2021)","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:59.115491Z"},"links":{"cited_paper":"/paper/2009.03300","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:e7564943c7498e34feff56b4c0fa56c1a662c517bb2feb36a163dcee363b058e","observation_id":"1f4eaf26-1294-4981-812f-a1f1136a981e","resolution":{"observed_at":"2026-08-07T15:26:59.115491Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14249","last_updated":"2026-02-20T04:23:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-24T05:27:46Z","title":"Humanity's Last Exam","version":10},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.14249","snapshot_observed_at":"2026-08-07T15:26:59.189173Z","title":"arXiv preprint arXiv:2501.14249 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:59.189173Z"},"links":{"cited_paper":"/paper/2501.14249","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:f6a0e5f2e2d08d185f9fbaac359b3b43e44e2a6ea7db9fa43c9cadad3a0fe293","observation_id":"78264078-8bb5-473a-a8db-9465570d5a75","resolution":{"observed_at":"2026-08-07T15:26:59.189173Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:59.256804Z","title":"https://scale.com/leaderboard/ humanitys_last_examAccessed: 2025-05-14","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:59.256804Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:669a1cf859d0ffb20d2d896c14ce3a19fc3fda1e75dc51246b6cdee46a5a5ca9","observation_id":"5215fa44-fcc9-4428-ac26-30c80fcaa06c","resolution":{"observed_at":"2026-08-07T15:26:59.256804Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:59.333175Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:59.333175Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:cf8bfb941000c141ec95d3ff734d7a35b4ecb8b5df4373b7aa950d8ea5b54021","observation_id":"fa21a2a7-b374-4ee5-8416-5e1e6fa90698","resolution":{"observed_at":"2026-08-07T15:26:59.333175Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.12022","last_updated":"2023-11-20T18:57:34Z","snapshot_observed_at":"2026-08-10T12:02:35.919497Z","submitted_at":"2023-11-20T18:57:34Z","title":"GPQA: A Graduate-Level Google-Proof Q&A Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.12022","snapshot_observed_at":"2026-08-07T15:26:59.405000Z","title":"arXiv preprint arXiv:2311.12022 (2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:59.405000Z"},"links":{"cited_paper":"/paper/2311.12022","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:cfbb5df192a224ddbabc18f474531bb2fef76ada747634ce8f05cf84d2db815e","observation_id":"05197398-9457-479b-8f9b-c55b05938893","resolution":{"observed_at":"2026-08-07T15:26:59.405000Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:59.464341Z","title":"https://www.vals.ai/benchmarks/ gpqa-05-09-2025Accessed: 2025-05-14","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:59.464341Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:dbc29304e1897f8b89241189109f2903b2e946c3dc9d2e08324c1f7d566d6ef1","observation_id":"18401cb3-ae82-436f-a734-1c1b8cf03368","resolution":{"observed_at":"2026-08-07T15:26:59.464341Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.16074","last_updated":"2025-05-18T14:13:34Z","snapshot_observed_at":"2026-08-12T23:17:53.534492Z","submitted_at":"2025-04-22T17:53:29Z","title":"PHYBench: Holistic Evaluation of Physical Perception and Reasoning in Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.16074","snapshot_observed_at":"2026-08-07T15:26:59.524310Z","title":"arXiv preprint arXiv:2504.16074 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:59.524310Z"},"links":{"cited_paper":"/paper/2504.16074","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:774e7d8c7d90a6397e6e69b38c00d1d47f69bcc240651695e317936ad1946b9c","observation_id":"c0b9f034-2e97-42c1-bb1a-d5525b541192","resolution":{"observed_at":"2026-08-07T15:26:59.524310Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.05444","last_updated":"2025-01-09T18:55:52Z","snapshot_observed_at":"2026-08-13T08:24:02.836306Z","submitted_at":"2025-01-09T18:55:52Z","title":"Can MLLMs Reason in Multimodality? EMMA: An Enhanced MultiModal ReAsoning Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.05444","snapshot_observed_at":"2026-08-07T15:26:59.566161Z","title":"arXiv preprint arXiv:2501.05444 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:59.566161Z"},"links":{"cited_paper":"/paper/2501.05444","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:94b6e63be64059d9991010172857478cbca131a360882d8e1b22b95a8158cb32","observation_id":"8cafd74e-8f58-4fde-a158-02ea21ec09c9","resolution":{"observed_at":"2026-08-07T15:26:59.566161Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:59.676392Z","title":"https://www.vals.ai/benchmarks/ math500-05-09-2025Accessed: 2025-05-14","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:59.676392Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:56c2e190c5f949594226d094febe467f2ad1f4e7a624b16cdb4238f0ef4f16e0","observation_id":"cce49f77-c6ae-4fb2-aab9-75b4ec1248d6","resolution":{"observed_at":"2026-08-07T15:26:59.676392Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:59.788838Z","title":"arXiv preprint arXiv:2410.03131 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:59.788838Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:0a2426bc277e6c19d73b948a1ab98a8b4528510b3f7795efded1d56a4556cf80","observation_id":"90b9f4e3-156d-4664-aee1-57fb91077fdf","resolution":{"observed_at":"2026-08-07T15:26:59.788838Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:59.975697Z","title":"https://www.vals.ai/benchmarks/ aime-2025-05-09Accessed: 2025-05-14","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:59.975697Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:6c15f89eccd8d05fa8ee6550dde4423af133bfee6bd30e568efcb5f6e77498e1","observation_id":"fd08dd1e-6fbf-48a1-98b3-78fd6a87369b","resolution":{"observed_at":"2026-08-07T15:26:59.975697Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.19314","last_updated":"2025-04-18T19:36:00Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-27T16:47:42Z","title":"LiveBench: A Challenging, Contamination-Limited LLM Benchmark","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.19314","snapshot_observed_at":"2026-08-07T15:27:00.152304Z","title":"arXiv preprint arXiv:2406.19314 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:00.152304Z"},"links":{"cited_paper":"/paper/2406.19314","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:baf12e01009654371a0132b0b3104fdeacca4e2750c7be0cc60118d5f033c244","observation_id":"b2b1ae1a-b149-46a6-8cf8-7810df23e33b","resolution":{"observed_at":"2026-08-07T15:27:00.152304Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:00.352797Z","title":"https://livebench.ai/#/?Coding=a& Mathematics=a&Data+Analysis=a&Language=a&IF=aAccessed: 2025-05-14","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:00.352797Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:03bb4fa09bc4d0719aeb24c5bc83ad92aaf5d4bd471b3e85a1d5a588878d7ef5","observation_id":"57ad3dd4-a046-4304-b642-c79511ba15da","resolution":{"observed_at":"2026-08-07T15:27:00.352797Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15877","last_updated":"2025-04-01T08:36:44Z","snapshot_observed_at":"2026-07-31T19:00:59.311189Z","submitted_at":"2024-06-22T15:52:04Z","title":"BigCodeBench: Benchmarking Code Generation with Diverse Function Calls and Complex Instructions","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15877","snapshot_observed_at":"2026-08-07T15:27:00.559765Z","title":"arXiv preprint arXiv:2406.15877 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:00.559765Z"},"links":{"cited_paper":"/paper/2406.15877","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:f2c13d3a9ef8f6f5344608b9acc34ccb51e59ae32554f99a93fb3939469d8294","observation_id":"57249563-1ef7-4de2-a8ef-873d95490edd","resolution":{"observed_at":"2026-08-07T15:27:00.559765Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:00.736996Z","title":"https://aider.chat/docs/leaderboards/ Ac- cessed: 2025-05-14","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:00.736996Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:e5d165dc233c29cff9f3cc1acffcc774a12aab9f7ea73f88843b2a308a5e3dfa","observation_id":"417548ee-d2f4-4b93-a41b-3768f88ef4a6","resolution":{"observed_at":"2026-08-07T15:27:00.736996Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:00.875924Z","title":"https://bigcode-bench.github.io/ Accessed: 2025-05-14","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:00.875924Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:cbe7b3feb20239be26e0c1f705f84954ad02ca847b5d6716f7bab92a4748b4f2","observation_id":"f74dbb8d-05cd-47d5-805c-429e6fb2df25","resolution":{"observed_at":"2026-08-07T15:27:00.875924Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:00.977620Z","title":"https://scale","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:00.977620Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:4e14364410dd77d023e4ee1d0f3c10c7443c2cd4852f609c3924e41b37f2b20d","observation_id":"5f981890-9fcb-428b-ba47-a854b244afe3","resolution":{"observed_at":"2026-08-07T15:27:00.977620Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:00.981839Z","title":"https://lmarena.ai/leaderboard Accessed: 2025-05-14","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:00.981839Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:6aa8da61d0756ed3ae419086be9bf851edd930903621f5fb2cbfd0f8d3a24ab9","observation_id":"1b3e83b5-e1af-4700-83e5-b546bd993e2a","resolution":{"observed_at":"2026-08-07T15:27:00.981839Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16502","last_updated":"2024-06-13T15:02:39Z","snapshot_observed_at":"2026-08-13T09:07:54.479155Z","submitted_at":"2023-11-27T17:33:21Z","title":"MMMU: A Massive Multi-discipline Multimodal Understanding and Reasoning Benchmark for Expert AGI","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.16502","snapshot_observed_at":"2026-08-07T15:27:00.986636Z","title":"arXiv preprint arXiv:2311.16502 (2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:00.986636Z"},"links":{"cited_paper":"/paper/2311.16502","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:f686743abbe142aee4a2c1ff06a428a35511b89af12e3d418fa2b80768697c89","observation_id":"c8b4e03f-6aa8-40d0-9580-048940736906","resolution":{"observed_at":"2026-08-07T15:27:00.986636Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:01.031667Z","title":"https://www.vals.ai/benchmarks/ mmmu-05-09-2025Accessed: 2025-05-14","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:01.031667Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:f98d279ab7088031c2e1c2aa9894f1770d022df45737640557fd91dfeffd74f2","observation_id":"46b6f099-93db-472a-bbb9-5991bf80b0b2","resolution":{"observed_at":"2026-08-07T15:27:01.031667Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17399","last_updated":"2025-03-06T04:41:56Z","snapshot_observed_at":"2026-08-10T04:40:42.053539Z","submitted_at":"2025-01-29T03:29:24Z","title":"MultiChallenge: A Realistic Multi-Turn Conversation Evaluation Benchmark Challenging to Frontier LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.17399","snapshot_observed_at":"2026-08-07T15:27:01.110876Z","title":"arXiv preprint arXiv:2501.17399 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:01.110876Z"},"links":{"cited_paper":"/paper/2501.17399","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:fe9bcd50bb1a572712f0e4e90a1e9c752d4cc4bb66946a529387886835c6eef1","observation_id":"8158aad3-a061-4172-a5e1-392d3e4fa3a4","resolution":{"observed_at":"2026-08-07T15:27:01.110876Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:01.208322Z","title":"https://scale.com/leaderboard/ multichallengeAccessed: 2025-05-14","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:01.208322Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:8cab7150651aa32bbaaaf1c8ee8c7e487ff2542bc0097bf3f4cd4f2fec2d9155","observation_id":"3e400887-44cc-4397-99cf-e6dcb36ebaa9","resolution":{"observed_at":"2026-08-07T15:27:01.208322Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:01.299976Z","title":"https://scale.com/leaderboard/ enigma_evalAccessed: 2025-05-14","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:01.299976Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:d2068b1620c0e0c6b4b0c00f2017bd288dae94d878d97c451aaf91fd1e152cb5","observation_id":"e543a205-cfd9-4a73-8081-570c10c774b0","resolution":{"observed_at":"2026-08-07T15:27:01.299976Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:01.362623Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:01.362623Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:e2d3909856424a270c360968ba117b29429911b98a022d3eacdd9b40399a9cf6","observation_id":"d1b43e71-9dfc-4115-b0db-dde5f55ad797","resolution":{"observed_at":"2026-08-07T15:27:01.362623Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:01.442781Z","title":"https://github.com/mpSchrader/gym-sokoban (2018)","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:01.442781Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:5434ccaf328a088ab438a7731bea67371731ba2ff6cb0d8bd3cf79474faaebcc","observation_id":"9b944751-7bca-4f85-a541-36dd67dd2b0b","resolution":{"observed_at":"2026-08-07T15:27:01.442781Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.620840Z","title":"https://github.com/jaybutera/ tetrisRL(2023) GitHub repository","venue":null,"work_id":"12cde6ce-d822-4595-840c-80ee4e52b193","year":2023},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:01.518032Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:f8fd65384072ed6df717db43f17eebbb29280ccdfdbdfc48573bdfa2ffe15efb","observation_id":"942416c6-c81b-4b08-a263-8b1b5a540611","resolution":{"observed_at":"2026-08-07T15:27:04.626893Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-07T15:27:01.523920Z","title":"5 technical report","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:01.523920Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:69db724d9cb852c88cb5712bc7afee7ed356224d568e433b560a43589c0c0813","observation_id":"086da939-2779-4dbc-9d8a-e528b44e84a9","resolution":{"observed_at":"2026-08-07T15:27:01.523920Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.604164Z","title":"Communications of the ACM 38(3) (1995) 58–68","venue":null,"work_id":"a1177d46-09d8-4309-813b-77c837d6615f","year":1995},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:01.528538Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:8e10c8362afa825cb0f50ed510f6cc4ebdf5cf43d999e0635ece69ee3230241e","observation_id":"cf891336-6b2c-40d8-abd9-19785f67c849","resolution":{"observed_at":"2026-08-07T15:27:04.609837Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.587428Z","title":"nature550(7676) (2017) 354–359","venue":null,"work_id":"c8f84a29-defd-4e52-b390-06db19078ba3","year":2017},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:01.544702Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:80fd42c8bcf6b8db6cb9e72329d3ef10dcfa62f8286a0923cf29b5c31a1e1e89","observation_id":"b0d4f444-27e6-4395-a8e0-5ae73574d782","resolution":{"observed_at":"2026-08-07T15:27:04.592570Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.571251Z","title":"In The Thirty-eight Conference on Neural Information Processing Systems Datasets and Benchmarks Track","venue":null,"work_id":"d1c0e405-701f-4faf-95c9-dbf239657d4f","year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:01.617277Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:3a6364cfb48a81aee92da81d6e0beaf37426c15c01331d472ab824dea77632c1","observation_id":"04da8a84-7fdb-4484-9b9a-ef622d288117","resolution":{"observed_at":"2026-08-07T15:27:04.576974Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09617","last_updated":"2025-03-06T20:13:02Z","snapshot_observed_at":"2026-08-12T18:48:17.834630Z","submitted_at":"2025-03-06T20:13:02Z","title":"Factorio Learning Environment","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09617","snapshot_observed_at":"2026-08-07T15:27:01.689046Z","title":"arXiv preprint arXiv:2503.09617 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:01.689046Z"},"links":{"cited_paper":"/paper/2503.09617","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:ebc3fe5269c0bc88735d79c142792575282de336e3a418213b600019f97f5bb0","observation_id":"f9065042-7769-4785-a445-20b67486def2","resolution":{"observed_at":"2026-08-07T15:27:01.689046Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.555948Z","title":"In The Thirty-eight Conference on Neural Information Processing Systems Datasets and Benchmarks Track","venue":null,"work_id":"ade258e3-23d9-4187-9990-05ce319dc3f8","year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:01.766369Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:42e7220de3d9b0ff2c6efddda4a0f4ba4507441093cc32206f5e0f0c48226dda","observation_id":"9a6fa030-734c-47ee-ad2d-e8c8af2764d1","resolution":{"observed_at":"2026-08-07T15:27:04.560892Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.18431","last_updated":"2025-02-25T18:26:48Z","snapshot_observed_at":"2026-08-07T17:49:22.453162Z","submitted_at":"2025-02-25T18:26:48Z","title":"TextGames: Learning to Self-Play Text-Based Puzzle Games via Language Model Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.18431","snapshot_observed_at":"2026-08-07T15:27:01.826741Z","title":"arXiv preprint arXiv:2502.18431 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:01.826741Z"},"links":{"cited_paper":"/paper/2502.18431","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:ed2ce2284a5104261a25de86e9c82a1923d84f37b662b69d20188bc4ca8d924d","observation_id":"16b11839-c133-4a36-bcd0-05ae76aa1262","resolution":{"observed_at":"2026-08-07T15:27:01.826741Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.10032","last_updated":"2023-08-19T14:33:40Z","snapshot_observed_at":"2026-08-14T05:13:13.994263Z","submitted_at":"2023-08-19T14:33:40Z","title":"GameEval: Evaluating LLMs on Conversational Games","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.10032","snapshot_observed_at":"2026-08-07T15:27:01.892670Z","title":"arXiv preprint arXiv:2308.10032 (2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:01.892670Z"},"links":{"cited_paper":"/paper/2308.10032","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:3c8665125f75c829705e031dbd22322cb062cffe6191044ef3498dd93d48e5b0","observation_id":"ef4dd5ca-d0f2-4830-a96b-26312327dea1","resolution":{"observed_at":"2026-08-07T15:27:01.892670Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.06394","last_updated":"2025-02-15T22:03:16Z","snapshot_observed_at":"2026-08-11T19:40:08.226596Z","submitted_at":"2024-12-09T11:22:59Z","title":"GameArena: Evaluating LLM Reasoning through Live Computer Games","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.06394","snapshot_observed_at":"2026-08-07T15:27:01.953231Z","title":"arXiv preprint arXiv:2412.06394 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:01.953231Z"},"links":{"cited_paper":"/paper/2412.06394","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:a6d92e633b1c12e78b40391d02a6e0b568658a58a25284e936a1e71fd40a1b6f","observation_id":"ff01265b-d150-4cf4-bf94-37cd01ea52a9","resolution":{"observed_at":"2026-08-07T15:27:01.953231Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06770","last_updated":"2024-11-11T23:05:04Z","snapshot_observed_at":"2026-08-12T14:56:35.025839Z","submitted_at":"2023-10-10T16:47:29Z","title":"SWE-bench: Can Language Models Resolve Real-World GitHub Issues?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06770","snapshot_observed_at":"2026-08-07T15:27:02.019566Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.019566Z"},"links":{"cited_paper":"/paper/2310.06770","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:8fbe7b5900ac109c91096ccb65a2449b9fc6286971d8f83899781355bbab75ac","observation_id":"6a8c9fc0-bf4f-4d18-a4ec-336252fdc380","resolution":{"observed_at":"2026-08-07T15:27:02.019566Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.13854","last_updated":"2024-04-16T15:13:18Z","snapshot_observed_at":"2026-08-06T12:47:01.809185Z","submitted_at":"2023-07-25T22:59:32Z","title":"WebArena: A Realistic Web Environment for Building Autonomous Agents","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.13854","snapshot_observed_at":"2026-08-07T15:27:02.059918Z","title":"arXiv preprint arXiv:2307.13854 (2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.059918Z"},"links":{"cited_paper":"/paper/2307.13854","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:35fb0619568456da3326587b0a42f6e9be3604a1b44f9ea7e9539bb932e67d94","observation_id":"93c0d409-ab27-49cb-8248-b6a7e5e39c44","resolution":{"observed_at":"2026-08-07T15:27:02.059918Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.13919","last_updated":"2024-06-06T18:37:34Z","snapshot_observed_at":"2026-08-09T21:47:22.232349Z","submitted_at":"2024-01-25T03:33:18Z","title":"WebVoyager: Building an End-to-End Web Agent with Large Multimodal Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.13919","snapshot_observed_at":"2026-08-07T15:27:02.065525Z","title":"arXiv preprint arXiv:2401.13919 (2024) 14","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.065525Z"},"links":{"cited_paper":"/paper/2401.13919","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:699d0a66b5506f144439ecd8b2f6890132519787e404199edfe48740b388b251","observation_id":"71003540-eb94-4f0a-8628-80fdb703865a","resolution":{"observed_at":"2026-08-07T15:27:02.065525Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.18901","last_updated":"2024-07-26T17:55:45Z","snapshot_observed_at":"2026-08-12T23:13:58.709084Z","submitted_at":"2024-07-26T17:55:45Z","title":"AppWorld: A Controllable World of Apps and People for Benchmarking Interactive Coding Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.18901","snapshot_observed_at":"2026-08-07T15:27:02.082501Z","title":"arXiv preprint arXiv:2407.18901 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.082501Z"},"links":{"cited_paper":"/paper/2407.18901","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:c8e96c2dd95f01bc071f6c59a53e8e3f6c29b6051ffb9ddcc225c83c143d1e0a","observation_id":"fbbce248-f0d4-44ce-9218-ff4a600c30a9","resolution":{"observed_at":"2026-08-07T15:27:02.082501Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.540766Z","title":"Advances in Neural Information Processing Systems37(2024) 52040–52094","venue":null,"work_id":"c5d0e6f1-ff42-4806-90dd-2443a043a1e5","year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.131440Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:116b1bae7121fdca1be8583c626a661dbb1ce4dd79beb9dd93d9180df7eb6fcd","observation_id":"55c10a9a-0b26-4173-92d4-352510af82cc","resolution":{"observed_at":"2026-08-07T15:27:04.545710Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.524038Z","title":null,"venue":null,"work_id":"0ec30dc5-50e6-439e-90d5-82b9dfa75fe5","year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.208401Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:c574804ff46e50aa336948ff284c8e4def036d03af4cf61b998e66b29380d616","observation_id":"23603768-7adc-4214-82f5-134ebcf65371","resolution":{"observed_at":"2026-08-07T15:27:04.528964Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.508562Z","title":"In The Twelfth International Conference on Learning Representations","venue":null,"work_id":"3f209b8d-ebae-4ea0-b234-eea563911356","year":2023},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.293383Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:39608c7253dcbbf0d6906ec5bddab5b0c311254dc49e185b919565cd0db8f9c0","observation_id":"ab6ef914-7e02-48fc-ac4c-3a0923f26339","resolution":{"observed_at":"2026-08-07T15:27:04.513351Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.490748Z","title":"Advances in neural information processing systems37(2024) 110935–110971","venue":null,"work_id":"e47933dd-7f2d-4b60-aec3-703d80ea4a66","year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.328494Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:2c3ddce7b80863d1250c4cab14de3429f5364a05ceaa46c357fe382606d4b5ba","observation_id":"ac079a74-203a-44ca-a45c-7e4e7b81e5a3","resolution":{"observed_at":"2026-08-07T15:27:04.497010Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-08-07T15:27:02.348919Z","title":"arXiv preprint arXiv:1707.06347 (2017)","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.348919Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:f2efb88de221fec016c9a3e52b971505a5b636f2c2b48940b7f8be63318bc5ef","observation_id":"dd33cdcf-25ab-4ac1-8e0a-508f1a56a9f0","resolution":{"observed_at":"2026-08-07T15:27:02.348919Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.473365Z","title":"Science10(3) (1995) 237–304","venue":null,"work_id":"4e8a52d5-7eb0-4c39-95ce-aa00449b0888","year":1995},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.397270Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:fced3352bfa50bd8d885e7b1673bd0cbbe8661ae84523235819902a23a345686","observation_id":"f23d2550-8302-4b52-9bb2-70b39c762a73","resolution":{"observed_at":"2026-08-07T15:27:04.478783Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.14476","last_updated":"2025-05-20T01:37:34Z","snapshot_observed_at":"2026-08-02T01:40:54.187278Z","submitted_at":"2025-03-18T17:49:06Z","title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.14476","snapshot_observed_at":"2026-08-07T15:27:02.445271Z","title":"arXiv preprint arXiv:2503.14476 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.445271Z"},"links":{"cited_paper":"/paper/2503.14476","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:a00e33ad0647adc36be6e9209b471b93361351d32b75bdc4b715375de8db8e0c","observation_id":"d17c69fb-0447-4821-94ee-8b7e972aff5c","resolution":{"observed_at":"2026-08-07T15:27:02.445271Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.456650Z","title":"Advances in Neural Information Processing Systems36(2023) 38975–38987","venue":null,"work_id":"5366aa45-3804-4012-8e2c-9620f6996297","year":2023},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.509094Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:eb088d98e7a1c9784d499b6c5fe3a0557fa94f379e4b1842301df7d780973931","observation_id":"bc3d9d72-699d-4e55-a952-6d4994acc0ff","resolution":{"observed_at":"2026-08-07T15:27:04.461822Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-14T02:43:01.480086Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-07T15:27:02.575808Z","title":"arXiv preprint arXiv:2110.14168 (2021)","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.575808Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:29e346695fa7b9a14ee2261fb92d53692e68225e2e3034d5a1d0c6bd3cd6b416","observation_id":"87ddd352-621e-4aab-a607-a901a790b5ca","resolution":{"observed_at":"2026-08-07T15:27:02.575808Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.438486Z","title":"Advances in Neural Information Processing Systems36(2024)","venue":null,"work_id":"adca9e0a-b4bb-47f0-aab7-090f16bd7a06","year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.602474Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:52440dfb06e0f4d8752998cd4e0d1e4e69f53ee93af812a2c4ef3ae1aca3d22f","observation_id":"102da256-cfb9-408f-8b4a-80014645a6b9","resolution":{"observed_at":"2026-08-07T15:27:04.444914Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.421210Z","title":"Advances in Neural Information Processing Systems 35(2022) 20744–20757","venue":null,"work_id":"79f1fb76-74a3-43fb-bf08-fb8278a01cc0","year":2022},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.606960Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:0794f95aca4a7abb9c200b8436656c783d43790017df368da06a9a4f4df54119","observation_id":"25bdbe78-156d-4464-be03-03384ac850ac","resolution":{"observed_at":"2026-08-07T15:27:04.426651Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.404265Z","title":"Biometrika30(1/2) (1938) 81–93","venue":null,"work_id":"c6f01a98-5400-42dc-8307-51d09f41cc9c","year":1938},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.679885Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:29371c85d3ccd9c44c74050f069a9f9605b71ea11c481a53bd77637a165b0daa","observation_id":"82404a63-dd5e-4dc3-bfea-47ec0b60f14c","resolution":{"observed_at":"2026-08-07T15:27:04.409726Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.387745Z","title":"In Proceedings of the 19th international conference on World wide web, ACM (2010) 577–586","venue":null,"work_id":"9cade593-f848-4223-a218-637af6e5cdb7","year":2010},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.740913Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:4fa32340e532ad79e14eb76c9d39789e1976ac44039ff579e810104d976fa5f4","observation_id":"49d0ee1f-ec40-4b19-b216-87c56e6c6c61","resolution":{"observed_at":"2026-08-07T15:27:04.392662Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.371875Z","title":"Educational Researcher5(10) (1976) 3–8","venue":null,"work_id":"40f93839-b076-46bb-b1da-12612a857309","year":1976},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.809560Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:4e6cbae1aade7c9a28b7e0873565e10a447fc544ec40acad5b5d9ccf5e9b3176","observation_id":"4f03dc3a-ea7e-47b1-8691-d7281719da5a","resolution":{"observed_at":"2026-08-07T15:27:04.376853Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.355536Z","title":"Block 1 is on top of block 3, block 3 is on top of block 2, and block 2 is on the table","venue":null,"work_id":"2ece6ea1-1eb5-4a85-b840-d35073e9d889","year":1908},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.852880Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:d45d483aaf3264083f55e6aaf72da8e17cb8cd0c016a3606110230a6123ad68e","observation_id":"fb942280-1434-4e92-8e42-927df2631386","resolution":{"observed_at":"2026-08-07T15:27:04.361377Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:02.911988Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.911988Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:33e2524a24d0d8432a89b812d1121b07c8771334a5861e62866d1cd41af20b84","observation_id":"616a293f-add6-4048-8061-6e2fee8b6378","resolution":{"observed_at":"2026-08-07T15:27:02.911988Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:02.950794Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.950794Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:dc0ab5f04bb99b30870fc86c64b7f0b7998f3c3db4a205800635e96db4ca58b2","observation_id":"9dda3d17-5a70-466d-8d56-55981e5b0ef1","resolution":{"observed_at":"2026-08-07T15:27:02.950794Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:03.032310Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:03.032310Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:a97e4091b35f60649388612643d58a8647dd1b07e6a50d2e8465ccff9650261f","observation_id":"e1830c7f-0d00-443b-9b70-32409578e2b0","resolution":{"observed_at":"2026-08-07T15:27:03.032310Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:03.119984Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:03.119984Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:79bda64b8e28952b9e0d0159611427a7f4dcd9acdd9f0c4c779b811128c64477","observation_id":"56d4f2d6-0d9a-4a9e-8129-55c3b56a358f","resolution":{"observed_at":"2026-08-07T15:27:03.119984Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.298615Z","title":"up\", \"down","venue":null,"work_id":"2730b75d-2248-4788-9fcf-70c81694822d","year":null},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:03.163759Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:cad4387a8ca4328b6f30902abd5aa5a7f1a49b00cf52e370c52521784ff6a433","observation_id":"fff924be-933e-4c74-bb34-f0059920b182","resolution":{"observed_at":"2026-08-07T15:27:04.304373Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:03.171436Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:03.171436Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:35b7935342922bbc4c89a6d92d13408055d53b53994abb227784bc4408398452","observation_id":"ce3e4880-c2cc-45a0-ae20-063acc143f72","resolution":{"observed_at":"2026-08-07T15:27:03.171436Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:03.177177Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:03.177177Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:68337c19f8c763f6a713b0c7ce2d68df932a2544611ba385c812cd06db22538f","observation_id":"f08e1eca-d190-4a28-b8c3-50d5a956d03c","resolution":{"observed_at":"2026-08-07T15:27:03.177177Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:03.182638Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":99,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:03.182638Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:a485400d1322831cb4dcba6729b9e8d068e3ba7ef65ca77d4729b136af7dc739","observation_id":"1025f4c9-f81b-4b4d-b541-07f78456c050","resolution":{"observed_at":"2026-08-07T15:27:03.182638Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:03.187728Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":100,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:03.187728Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:9fb7148afcc9cdde0b5ae2b5f37ff2c47e6fa347fe5f1f92366ec1a1ece9c828","observation_id":"c58fb195-9556-4c8d-ae4b-178daf346b75","resolution":{"observed_at":"2026-08-07T15:27:03.187728Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","latest_version":2,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":1,"metadata_mismatch":2,"parse_uncertain":0,"unresolved":81,"verified_exact":0,"verified_fuzzy":16},"total_outbound_references":115},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"thesis":"As of 14 August 2026, this Paper Citation Record lists 100 of 115 outbound references and 22 inbound Pith citation observations for arXiv:2505.15146."}