{"as_of":"2026-08-11T13:56:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e0c21333eff1b60e145b98304eb3e26e14b7e3b9025379be1dcd6a50e44105bf","coverage":[{"denominator":42,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":42,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T14:23:27.591640Z","state":"measured"},{"denominator":44,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":44,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T23:34:01.106317Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"cited_work":{"arxiv_id":"2508.21365","doi":"10.48550/arxiv.2508.21365","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.21365","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Think in games: Learning to reason in games via reinforcement learning with large language models,","venue":"ArXiv.org","work_id":"84af43b1-3481-4db6-9574-7bece161ed5a","year":2025},"citing_paper":{"arxiv_id":"2606.12191","last_updated":"2026-06-10T15:15:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-06-10T15:15:01Z","title":"Agentic Environment Engineering for Large Language Models: A Survey of Environment Modeling, Synthesis, Evaluation, and Application","version":1},"reference_index":288,"source":"pdf_text","source_observed_at":"2026-06-27T09:46:30.702256Z"},"links":{"cited_paper":"/paper/2508.21365","citing_paper":"/paper/2606.12191"},"observation_digest":"sha256:02b25cfc26b036bf1f9131d910423030861c3ec50ea68f5974c26df9ae81a3a1","observation_id":"682ba848-7a4b-4e77-a5d3-b9c69669f7e6","resolution":{"observed_at":"2026-06-27T09:50:48.309200Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.21365","snapshot_observed_at":"2026-08-04T23:34:01.106317Z","title":"arXiv preprint arXiv:2508.21365 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.01652","last_updated":"2026-08-03T03:43:57Z","snapshot_observed_at":"2026-08-10T04:13:40.745631Z","submitted_at":"2026-08-03T03:43:57Z","title":"SyncPlan: Long-Horizon LLM Coordination with Explicit Synchronization and Adaptive Correction","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-04T23:34:01.106317Z"},"links":{"cited_paper":"/paper/2508.21365","citing_paper":"/paper/2608.01652"},"observation_digest":"sha256:a6345f013235131eae8a53239aa5d9c3be183d176efc9ed9c34ad880b4bac183","observation_id":"f1e7d76b-4a07-474e-8ef8-18cc3018c509","resolution":{"observed_at":"2026-08-04T23:34:01.106317Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2508.21365/citation-record","integrity":"/paper/2508.21365/integrity","json":"/paper/2508.21365/citation-record.json","paper":"/paper/2508.21365"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2402.18139","last_updated":"2024-09-30T00:40:00Z","snapshot_observed_at":"2026-08-08T06:38:37.632531Z","submitted_at":"2024-02-28T08:02:14Z","title":"Cause and Effect: Can Large Language Models Truly Understand Causality?","version":3},"cited_work":{"arxiv_id":"2402.18139","doi":"10.48550/arxiv.2402.18139","metadata_source":"pith","pith_arxiv_id":"2402.18139","snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"Cause and Effect: Can Large Language Models Truly Understand Causality?","venue":"cs.CL","work_id":"4fd48f01-b5cb-4e6c-bcad-6b4271d22318","year":2024},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:23.843562Z"},"links":{"cited_paper":"/paper/2402.18139","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:75ecbfa15abd492bdf34e2db1b540b72a9814efe70236e80b2c2ebd502494d23","observation_id":"3462f19f-6220-49a1-a503-049fa24058ce","resolution":{"observed_at":"2026-08-05T14:23:28.247417Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:23:23.929458Z","title":null,"venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:23.929458Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:32adec8ae669031ba98666639a9b327400e0d178d6e611927bfd4e113a81a5ef","observation_id":"6039e9bf-ca8c-441f-97b9-581a2cd2d618","resolution":{"observed_at":"2026-08-05T14:23:23.929458Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-05T14:23:24.020818Z","title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:24.020818Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:bfb840bf67abee99c8bd615e1d29b996939dcf8dbf2973bf15b1cc7b9bdb3f91","observation_id":"9b2c6e68-646a-43be-92f7-971cf3f8ff64","resolution":{"observed_at":"2026-08-05T14:23:24.020818Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.00316","last_updated":"2025-06-06T08:14:05Z","snapshot_observed_at":"2026-08-10T22:51:29.177247Z","submitted_at":"2024-12-31T07:20:32Z","title":"MapEval: A Map-Based Evaluation of Geo-Spatial Reasoning in Foundation Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.00316","snapshot_observed_at":"2026-08-05T14:23:24.072935Z","title":"Mapeval: A map-based evaluation of geo-spatial reasoning in foundation models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:24.072935Z"},"links":{"cited_paper":"/paper/2501.00316","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:b35eec6b4768c1c2181a1fca4411b25550cb8012f28f8b6ccacfaa9ffa0bda5c","observation_id":"65eeae8c-36e2-48ce-8939-8ceb51a810c7","resolution":{"observed_at":"2026-08-05T14:23:24.072935Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1016/j.patrec.2007.06.013","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"Bayeschess: A computer chess program based on bayesian networks","venue":"Pattern Recognition Letters","work_id":"f2949566-7044-482a-91a5-03d0a02fc0c7","year":2008},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:24.156489Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:8b8777e402e80ea3ddfbf342b3e3785122502fe93ddd2bb20b56276f14582012","observation_id":"7a428987-6583-4222-9603-fb9898fd35bc","resolution":{"observed_at":"2026-08-05T14:23:28.080495Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2018.28345","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:23:29.842234Z","title":"Font and Tobias Mahlmann","venue":null,"work_id":"36499384-e109-4541-a224-1b7bf3044b36","year":2019},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:24.207525Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:968a0d7c411669b81295c645cdbb84d645b5ba2a1056f1bf29253360d80bc590","observation_id":"d4b144ae-8637-4860-883c-01ca553fd71a","resolution":{"observed_at":"2026-08-05T14:23:29.926244Z","resolver_source":"raw_fallback","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2507.17131","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:23:29.571722Z","title":"Enabling self-improving agents to learn at test time with human-in-the-loop guidance","venue":null,"work_id":"187121ff-cbb7-4f66-a256-cad4d13661bd","year":2025},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:24.271531Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:ea3413f57b4060b55170a5abd693ba7574d4a0f2aa75baca04f56f2477e181a2","observation_id":"ce16c2fa-817c-4f36-9a6e-3bc437a6a1b5","resolution":{"observed_at":"2026-08-05T14:23:29.665817Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:23:24.347757Z","title":"Measuring massive multitask language understanding","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:24.347757Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:148dd8b8460ef0f58ebdbd9cd28d05e3e29d6fd8766b7d7529c570856e94c1f0","observation_id":"7bb921e6-e04c-48b4-83d7-f04eef122fb0","resolution":{"observed_at":"2026-08-05T14:23:24.347757Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.11143","last_updated":"2025-10-09T12:22:46Z","snapshot_observed_at":"2026-07-31T12:28:37.704994Z","submitted_at":"2024-05-20T01:04:40Z","title":"OpenRLHF: An Easy-to-use, Scalable and High-performance RLHF Framework","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.11143","snapshot_observed_at":"2026-08-05T14:23:24.434666Z","title":"Openrlhf: An easy-to-use, scalable and high-performance rlhf framework","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:24.434666Z"},"links":{"cited_paper":"/paper/2405.11143","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:4bf1f4fd5dea10c645e86b2d603387c74a5bb1faeb3d9eb54d5edc3de004928e","observation_id":"712a2b16-a84f-48c1-95db-a3c78fcec02a","resolution":{"observed_at":"2026-08-05T14:23:24.434666Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.02039","last_updated":"2026-06-08T03:25:04Z","snapshot_observed_at":"2026-08-03T07:51:57.376334Z","submitted_at":"2024-04-02T15:34:18Z","title":"A Survey on Large Language Model-Based Game Agents","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.02039","snapshot_observed_at":"2026-08-05T14:23:24.473717Z","title":"A survey on large language model-based game agents","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:24.473717Z"},"links":{"cited_paper":"/paper/2404.02039","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:541189ab72e334efd942d0287cd8595053d6b75528f6f3f4fe51e8002f130d34","observation_id":"4d97786b-a523-4a9f-a8cb-623b7b38872f","resolution":{"observed_at":"2026-08-05T14:23:24.473717Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01118","last_updated":"2024-04-02T15:46:35Z","snapshot_observed_at":"2026-08-06T08:24:17.984186Z","submitted_at":"2024-02-02T03:22:12Z","title":"PokeLLMon: A Human-Parity Agent for Pokemon Battles with Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01118","snapshot_observed_at":"2026-08-05T14:23:24.541597Z","title":"Pokellmon: A human-parity agent for pokemon battles with large language models, 2024 c","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:24.541597Z"},"links":{"cited_paper":"/paper/2402.01118","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:ba4ccf60bdd074fba8d67f6a0909b6a154e33c426bdd5a20342495b5d5d44179","observation_id":"37e15f8e-9e20-42f3-b515-e38236208f3f","resolution":{"observed_at":"2026-08-05T14:23:24.541597Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:23:30.823398Z","title":"C-eval: A multi-level multi-discipline chinese evaluation suite for foundation models","venue":null,"work_id":"f965f9f5-fd02-427e-9bea-6eacf7746590","year":2023},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:24.602432Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:16de17d5b222738f3e322067bb3072f48c1ab83f282ab5f4c8275ebbc4c12d21","observation_id":"b120537f-e788-4299-b64b-8bc3f74f543e","resolution":{"observed_at":"2026-08-05T14:23:30.920211Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09516","last_updated":"2025-08-05T19:08:38Z","snapshot_observed_at":"2026-07-06T20:51:28.022519Z","submitted_at":"2025-03-12T16:26:39Z","title":"Search-R1: Training LLMs to Reason and Leverage Search Engines with Reinforcement Learning","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09516","snapshot_observed_at":"2026-08-05T14:23:24.667850Z","title":"Search-r1: Training llms to reason and leverage search engines with reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:24.667850Z"},"links":{"cited_paper":"/paper/2503.09516","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:a7a69b6c6e9195bdba2a1e05da998df3033813edf6a224902f6d9d0916ab1e35","observation_id":"0a47fa39-e61c-4bd8-90a1-6ee078b5988d","resolution":{"observed_at":"2026-08-05T14:23:24.667850Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.20213","last_updated":"2025-04-01T14:43:55Z","snapshot_observed_at":"2026-08-06T03:54:57.091503Z","submitted_at":"2024-09-30T11:48:11Z","title":"Mind the GAP: Glimpse-based Active Perception improves generalization and sample efficiency of visual reasoning","version":2},"cited_work":{"arxiv_id":"2409.20213","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.20213","snapshot_observed_at":"2026-08-05T14:23:29.328001Z","title":"Mind the GAP: Glimpse-based Active Perception improves generalization and sample efficiency of visual reasoning","venue":"cs.CV","work_id":"4262f029-97b9-4f90-b5e6-2a6e7493ffdd","year":2024},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:24.726570Z"},"links":{"cited_paper":"/paper/2409.20213","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:fd1c127fbba65a8695982e0cf98756badcbeb61a36665d8a4b16a37be6ba0e55","observation_id":"b8d24579-c921-478d-a8ca-613a937f29ae","resolution":{"observed_at":"2026-08-05T14:23:29.386562Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:23:30.683941Z","title":"School chinese benchmark, 2018","venue":null,"work_id":"4d4d20d1-1e19-43bc-950f-d2194a7c8b99","year":2018},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:24.798984Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:220e2c9eda2bdcfd5f8dc1ae2837dbffb4bb678a63038a86553309b60c2a22fb","observation_id":"a7266d34-80bb-40ec-a956-b4b069ea27d6","resolution":{"observed_at":"2026-08-05T14:23:30.727816Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.07316","last_updated":"2025-05-21T13:38:27Z","snapshot_observed_at":"2026-08-10T03:48:50.081958Z","submitted_at":"2025-02-11T07:26:50Z","title":"CodeI/O: Condensing Reasoning Patterns via Code Input-Output Prediction","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.07316","snapshot_observed_at":"2026-08-05T14:23:24.911042Z","title":"Codei/o: Condensing reasoning patterns via code input-output prediction","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:24.911042Z"},"links":{"cited_paper":"/paper/2502.07316","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:c621b95ea9c9e124ea04c23b6d679ace4393894e8190d7082d605390e077460d","observation_id":"4d0d153b-c670-4065-85c6-183d48962969","resolution":{"observed_at":"2026-08-05T14:23:24.911042Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:23:24.975898Z","title":"Simpo: Simple preference optimization with a reference-free reward","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:24.975898Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:0a619942b92d40376f36a3ecee3540e52372b427ce150fae6b44961d369882db","observation_id":"d0e6f647-06ac-422b-be93-43d35f0c9b2e","resolution":{"observed_at":"2026-08-05T14:23:24.975898Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1312.5602","last_updated":"2013-12-19T16:00:08Z","snapshot_observed_at":"2026-07-06T03:31:23.521122Z","submitted_at":"2013-12-19T16:00:08Z","title":"Playing Atari with Deep Reinforcement Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1312.5602","snapshot_observed_at":"2026-08-05T14:23:25.071210Z","title":"Playing atari with deep reinforcement learning","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:25.071210Z"},"links":{"cited_paper":"/paper/1312.5602","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:a3bac5e3c783198393e6dc350f7a2f40f88e061bd9cc2b59cbc6bb2b426b19fb","observation_id":"8fe91552-3333-4fac-b23d-823e54708096","resolution":{"observed_at":"2026-08-05T14:23:25.071210Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2021.30495","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:23:29.059062Z","title":"Creating pro-level AI for a real-time fighting game using deep reinforcement learning","venue":null,"work_id":"0ef64497-6163-4874-b91d-11fd6c28cb77","year":2022},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:25.134829Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:01ce2358e3009f19d94ba4c5f357027e696cdb4accc52a10ae77a9299f19332a","observation_id":"2ad2ca66-1586-4690-931c-d94db8641ab5","resolution":{"observed_at":"2026-08-05T14:23:29.160079Z","resolver_source":"raw_fallback","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:23:30.518999Z","title":"Wainwright, Pamela Mishkin, Chong Zhang, Sandhini Agarwal, Katarina Slama, et al","venue":null,"work_id":"779fda87-dffb-4bcf-b5ad-b8dd8e821487","year":2022},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:25.311280Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:10243deef75f65ca62515415d5f363af3e2c8a2b3130e6d872135609cc53a955","observation_id":"30ed4bd9-ea79-4473-95b0-6d637aab7769","resolution":{"observed_at":"2026-08-05T14:23:30.589098Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:23:25.379171Z","title":"Manning, Stefano Ermon, and Chelsea Finn","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:25.379171Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:8c346dd69e9e306f942808c9edaf6f7de3c7ac91f3713ba449826b9b0622f352","observation_id":"cf1e4c3f-0971-4329-965f-2e2f5a9a7df6","resolution":{"observed_at":"2026-08-05T14:23:25.379171Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-08-05T14:23:25.463390Z","title":"Proximal policy optimization algorithms","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:25.463390Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:70c8f8eeba4cd9c8e4c6d1b911fc7de9450bac993a786307c94d2febcf2f1b34","observation_id":"95e7b3ed-3fcd-47db-89ce-a6f1e284b427","resolution":{"observed_at":"2026-08-05T14:23:25.463390Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-05T14:23:25.587387Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:25.587387Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:9c5d45f6dbdb93fe3dc7878ccbb3c1282d4d0126d8e2da407d99e63eadbec523","observation_id":"6ad11b99-f8f3-4a63-b322-a6221a23776a","resolution":{"observed_at":"2026-08-05T14:23:25.587387Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1909.08053","last_updated":"2020-03-13T23:45:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-09-17T19:42:54Z","title":"Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.08053","snapshot_observed_at":"2026-08-05T14:23:25.667622Z","title":"Megatron-lm: Training multi-billion parameter language models using model parallelism","venue":null,"work_id":null,"year":1909},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:25.667622Z"},"links":{"cited_paper":"/paper/1909.08053","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:da224c8c81cae5adfdff8663f2858187786617ff3be903ecef2bbaa7c90798b7","observation_id":"5678c3e7-8d6f-494f-b41c-ad52a791efca","resolution":{"observed_at":"2026-08-05T14:23:25.667622Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:23:25.840989Z","title":"Mastering the game of go with deep neural networks and tree search","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:25.840989Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:979096f048bea888528fb8934d723c98f753fcd72268420078c44da94ef5e74a","observation_id":"c3e1d71b-d402-466c-a341-5baf90e30292","resolution":{"observed_at":"2026-08-05T14:23:25.840989Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1207.1411","last_updated":"2012-07-04T16:22:47Z","snapshot_observed_at":"2026-07-06T02:51:20.159654Z","submitted_at":"2012-07-04T16:22:47Z","title":"Bayes' Bluff: Opponent Modelling in Poker","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1207.1411","snapshot_observed_at":"2026-08-05T14:23:26.017512Z","title":"Bayes' bluff: Opponent modelling in poker","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:26.017512Z"},"links":{"cited_paper":"/paper/1207.1411","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:10dc5e22cf8d65a57cd0be8d2e83294939ae2f128facb146ea3afe78e2610fc3","observation_id":"a01d5683-b46c-40b4-9454-fdbea02541dc","resolution":{"observed_at":"2026-08-05T14:23:26.017512Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:23:30.336552Z","title":"Brown, Adam Santoro, Aditya Gupta, et al","venue":null,"work_id":"62cf0502-18fe-421b-8173-982156eb4b40","year":2023},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:26.131722Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:1916f6027e55c9293494578cc368d55070f7c3168e41f5f6c06444a842743e7f","observation_id":"2eb77aa2-f1af-4a39-8bb1-bc87abb3f955","resolution":{"observed_at":"2026-08-05T14:23:30.402090Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19918","last_updated":"2026-05-07T16:58:35Z","snapshot_observed_at":"2026-08-02T05:40:27.766263Z","submitted_at":"2025-02-27T09:40:13Z","title":"Meta-Reasoner: Dynamic Guidance for Optimized Inference-time Reasoning in Large Language Models","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.19918","snapshot_observed_at":"2026-08-05T14:23:26.207323Z","title":"Meta-reasoner: Dynamic guidance for optimized inference-time reasoning in large language models, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:26.207323Z"},"links":{"cited_paper":"/paper/2502.19918","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:228c184347575d34fcfc801bfe324e4910e32e82746134ee73e039310af268ba","observation_id":"8c0b0d61-6407-4193-b951-8fee35869ffe","resolution":{"observed_at":"2026-08-05T14:23:26.207323Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:23:26.320509Z","title":"Le, Ed H","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:26.320509Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:4f91848fa5d32389e1ddf4220e98fc0bc6e9daf0612dfb47e74f8142235a7e1e","observation_id":"ad54ce16-f332-43d9-82dd-ddde84dc1c39","resolution":{"observed_at":"2026-08-05T14:23:26.320509Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.01275","last_updated":"2024-01-09T18:54:05Z","snapshot_observed_at":"2026-08-08T19:26:30.965589Z","submitted_at":"2024-01-02T16:20:40Z","title":"CharacterEval: A Chinese Benchmark for Role-Playing Conversational Agent Evaluation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.01275","snapshot_observed_at":"2026-08-05T14:23:26.402477Z","title":"Charactereval: A chinese benchmark for role-playing conversational agent evaluation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:26.402477Z"},"links":{"cited_paper":"/paper/2401.01275","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:4b94cdeac3cd111e992aa9d262e59c07511332e67aad3fe6364e3e12151f43bb","observation_id":"7cb96572-77c7-4a03-bca6-b78143d72d34","resolution":{"observed_at":"2026-08-05T14:23:26.402477Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1708.04782","last_updated":"2017-08-16T06:20:52Z","snapshot_observed_at":"2026-08-10T10:13:54.999657Z","submitted_at":"2017-08-16T06:20:52Z","title":"StarCraft II: A New Challenge for Reinforcement Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1708.04782","snapshot_observed_at":"2026-08-05T14:23:26.486476Z","title":"Starcraft ii: A new challenge for reinforcement learning, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:26.486476Z"},"links":{"cited_paper":"/paper/1708.04782","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:18c4ed878a2516b113fb72aa5d1ef4f9a8f864f25f5234f0b7da518a56a3fa1b","observation_id":"e1d6f15b-32c1-4415-95b7-cce9431820eb","resolution":{"observed_at":"2026-08-05T14:23:26.486476Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.16291","last_updated":"2023-10-19T16:27:03Z","snapshot_observed_at":"2026-08-07T08:29:46.650400Z","submitted_at":"2023-05-25T17:46:38Z","title":"Voyager: An Open-Ended Embodied Agent with Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.16291","snapshot_observed_at":"2026-08-05T14:23:26.587398Z","title":"Voyager: An open-ended embodied agent with large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:26.587398Z"},"links":{"cited_paper":"/paper/2305.16291","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:b19115e77494a58cee8ef563542610d2445ad02f377f6ccc675c02de4f154d5c","observation_id":"6ec25b2f-cfd9-4b6c-a1f6-3fa2eeb994c4","resolution":{"observed_at":"2026-08-05T14:23:26.587398Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.01560","last_updated":"2024-07-08T05:56:47Z","snapshot_observed_at":"2026-08-02T14:50:04.461434Z","submitted_at":"2023-02-03T06:06:27Z","title":"Describe, Explain, Plan and Select: Interactive Planning with Large Language Models Enables Open-World Multi-Task Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.01560","snapshot_observed_at":"2026-08-05T14:23:26.680923Z","title":"Describe, explain, plan and select: Interactive planning with large language models enables open-world multi-task agents, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:26.680923Z"},"links":{"cited_paper":"/paper/2302.01560","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:66170230c992dfe24d8309c0db95578b353f78505fb6e8c55c0643c5da61ea79","observation_id":"6ef93714-881f-4b35-80d7-bf99c4f62fd2","resolution":{"observed_at":"2026-08-05T14:23:26.680923Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03622","last_updated":"2024-10-23T07:20:26Z","snapshot_observed_at":"2026-08-01T23:41:06.358986Z","submitted_at":"2024-04-04T17:45:08Z","title":"Mind's Eye of LLMs: Visualization-of-Thought Elicits Spatial Reasoning in Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.03622","snapshot_observed_at":"2026-08-05T14:23:26.747722Z","title":"Mind's eye of llms: Visualization-of-thought elicits spatial reasoning in large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:26.747722Z"},"links":{"cited_paper":"/paper/2404.03622","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:b8d73d054a277d3ded6a830b18e8c7e09c7aef1f9128390863d0b6210c239114","observation_id":"1f538ca5-df5c-4728-8f1e-c26e3ae4beff","resolution":{"observed_at":"2026-08-05T14:23:26.747722Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.13356","last_updated":"2025-03-17T16:42:34Z","snapshot_observed_at":"2026-08-07T16:57:17.977712Z","submitted_at":"2025-03-17T16:42:34Z","title":"Agents Play Thousands of 3D Video Games","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.13356","snapshot_observed_at":"2026-08-05T14:23:26.797595Z","title":"Agents play thousands of 3d video games","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:26.797595Z"},"links":{"cited_paper":"/paper/2503.13356","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:6defdcb5747d36f5163e4118716eaf6a7f9b75f55f6cd1c49d58a60bd3a2d06a","observation_id":"e9e3842a-54be-4d53-beb0-b66881e83778","resolution":{"observed_at":"2026-08-05T14:23:26.797595Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2502.12530","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:23:28.496152Z","title":"Policy-to-language: Train llms to explain decisions with flow-matching generated rewards","venue":null,"work_id":"9a83da07-05eb-42bf-be78-b9f892334e2f","year":2025},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:26.982264Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:131d36e8b7baa32c777f41ac0b12e4d8a9d6296eadcd81c6142395035318aa19","observation_id":"5dbef2c2-1c83-443d-93f3-8b5e6f3c6e76","resolution":{"observed_at":"2026-08-05T14:23:28.595131Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:23:30.188680Z","title":"Mastering complex control in moba games with deep reinforcement learning","venue":null,"work_id":"1da67548-d32d-422c-b88e-f190ac901258","year":2020},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:27.199232Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:22798c995b7a1f83c339e6485922442f2d16f22b1dadd123519aaffb1acdbbbc","observation_id":"e6705ad7-3bbe-4ea4-a259-aaef503742f2","resolution":{"observed_at":"2026-08-05T14:23:30.251013Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2024.acl-demos.30","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"C har P oet: A C hinese classical poetry generation system based on token-free LLM","venue":null,"work_id":"9f257eaa-ff5a-41b6-9e25-a1c595e2498b","year":2024},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:27.276998Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:975eb71bc581baf24cb08a4d753aba13553db7fb48ef63986dec12828b303e10","observation_id":"34de5102-2fc4-47ed-a428-202845f6c721","resolution":{"observed_at":"2026-08-05T14:23:27.806325Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:23:30.036058Z","title":"Training interactive agent in large fps game map with rule-enhanced reinforcement learning","venue":null,"work_id":"d0a8b54b-d910-4c23-ba80-e1b8ca0c5d45","year":2024},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:27.367053Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:aa8b0f88e6fd403cfe5f2cc4ba8af4cdf56470b69d176b5e7058c9d092a04797","observation_id":"5a7d02ff-d198-40fd-88b0-c455de45c09f","resolution":{"observed_at":"2026-08-05T14:23:30.097095Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2009.11506","last_updated":"2020-10-09T01:36:35Z","snapshot_observed_at":"2026-07-06T09:58:22.699198Z","submitted_at":"2020-09-24T06:17:10Z","title":"Ape210K: A Large-Scale and Template-Rich Dataset of Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.11506","snapshot_observed_at":"2026-08-05T14:23:27.444486Z","title":"Ape210k: A large-scale and template-rich dataset of math word problems, 2020","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:27.444486Z"},"links":{"cited_paper":"/paper/2009.11506","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:66a2e798a9ce9cdb0901bd607701fce5fdbda4798f536257bc9c3dda1b914356","observation_id":"0aeb653d-cc59-49f5-ac7c-a1e24532cf66","resolution":{"observed_at":"2026-08-05T14:23:27.444486Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07911","last_updated":"2023-11-14T05:13:55Z","snapshot_observed_at":"2026-07-06T16:47:08.877195Z","submitted_at":"2023-11-14T05:13:55Z","title":"Instruction-Following Evaluation for Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.07911","snapshot_observed_at":"2026-08-05T14:23:27.530464Z","title":"Instruction-following evaluation for large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:27.530464Z"},"links":{"cited_paper":"/paper/2311.07911","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:5755ba89c755030362ed0b777ca2850f40e0bbeee2e89563b73ea88f308528b9","observation_id":"4007ec55-0812-434b-9b84-7b42004e4305","resolution":{"observed_at":"2026-08-05T14:23:27.530464Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08328","last_updated":"2025-01-24T20:15:10Z","snapshot_observed_at":"2026-08-10T20:25:30.236105Z","submitted_at":"2025-01-14T18:59:03Z","title":"PokerBench: Training Large Language Models to become Professional Poker Players","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.08328","snapshot_observed_at":"2026-08-05T14:23:27.591640Z","title":"Pokerbench: Training large language models to become professional poker players","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:27.591640Z"},"links":{"cited_paper":"/paper/2501.08328","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:525a2283d0bb9202e16dc996411461225945a69af7813027486fdfcaba00d654","observation_id":"9f49b6f6-5ae9-4788-8807-a56fc5849d5e","resolution":{"observed_at":"2026-08-05T14:23:27.591640Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-07T23:56:02.972428Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models"},"reference_resolution":{"displayed":42,"state_counts":{"malformed_identifier":1,"metadata_mismatch":2,"parse_uncertain":0,"unresolved":27,"verified_exact":6,"verified_fuzzy":6},"total_outbound_references":42},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 42 of 42 outbound references and 2 inbound Pith citation observations for arXiv:2508.21365."}