{"as_of":"2026-08-18T07:22:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:cc996839f43e3caec105400745fa4a22818e4a52c5506e5bd3d1f3d9b7b8c3eb","coverage":[{"denominator":56,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":56,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T18:44:00.020826Z","state":"measured"},{"denominator":57,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":57,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-18T06:34:40.430872+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-12T08:40:40.910461Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-12T08:40:41.143168Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"cited_work":{"arxiv_id":"2507.07498","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.07498","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Teaching llm to reason: Reinforcement learning from algorithmic problems without code","venue":null,"work_id":"bf18b58f-d5fb-46df-8813-c1a01e69c402","year":2025},"citing_paper":{"arxiv_id":"2503.09567","last_updated":"2025-07-18T15:57:54Z","snapshot_observed_at":"2026-08-08T22:33:20.124926Z","submitted_at":"2025-03-12T17:35:03Z","title":"Towards Reasoning Era: A Survey of Long Chain-of-Thought for Reasoning Large Language Models","version":5},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-12T08:40:40.910461Z"},"links":{"cited_paper":"/paper/2507.07498","citing_paper":"/paper/2503.09567"},"observation_digest":"sha256:92efbb14874efd68cdacd4f9da75c777b744385a1ca1146ac71324010db4ce0c","observation_id":"bd083f39-b368-43e4-831b-cbe8ff80f22e","resolution":{"observed_at":"2026-05-12T08:40:41.146090Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2507.07498/citation-record","integrity":"/paper/2507.07498/integrity","json":"/paper/2507.07498/citation-record.json","paper":"/paper/2507.07498"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2108.07732","last_updated":"2021-08-16T03:57:30Z","snapshot_observed_at":"2026-08-15T17:40:38.050939Z","submitted_at":"2021-08-16T03:57:30Z","title":"Program Synthesis with Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2108.07732","snapshot_observed_at":"2026-08-06T18:43:56.513180Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:56.513180Z"},"links":{"cited_paper":"/paper/2108.07732","citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:f0f9c0ae3700168ca9864ca619e9c296d259a731d851ba77efed8dce86ce1909","observation_id":"e97d3ca6-0613-403f-a167-1f9d2740b4e0","resolution":{"observed_at":"2026-08-06T18:43:56.513180Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2107.03374","last_updated":"2021-07-14T17:16:02Z","snapshot_observed_at":"2026-08-08T11:58:24.516369Z","submitted_at":"2021-07-07T17:41:24Z","title":"Evaluating Large Language Models Trained on Code","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2107.03374","snapshot_observed_at":"2026-08-06T18:43:56.595227Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:56.595227Z"},"links":{"cited_paper":"/paper/2107.03374","citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:f86aa912b3d4088f1d5ce902310f8c81064602e98505a0a6c737736982d95883","observation_id":"3498fbf5-206a-4956-b0ee-4d4b3abd7f41","resolution":{"observed_at":"2026-08-06T18:43:56.595227Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09567","last_updated":"2025-07-18T15:57:54Z","snapshot_observed_at":"2026-08-08T22:33:20.124926Z","submitted_at":"2025-03-12T17:35:03Z","title":"Towards Reasoning Era: A Survey of Long Chain-of-Thought for Reasoning Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09567","snapshot_observed_at":"2026-08-06T18:43:56.693037Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:56.693037Z"},"links":{"cited_paper":"/paper/2503.09567","citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:707ea6f190f5cbb7b2d78a94ef40dffc4e13580293af6e0990e88531d1861dc8","observation_id":"f7db6b0e-5d22-48a2-b67e-999a3b922f9f","resolution":{"observed_at":"2026-08-06T18:43:56.693037Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:44:01.654018Z","title":null,"venue":null,"work_id":"4c937593-29ed-41c1-a190-bd482a5394aa","year":2025},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:56.789538Z"},"links":{"citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:7123441d1090cc8610557cda2ce6fda134ce4bb9ab2dc3f93e070d42cc5a91d4","observation_id":"f3e74ca6-911d-4584-919d-901d130cc690","resolution":{"observed_at":"2026-08-06T18:44:01.702849Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17161","last_updated":"2025-05-26T17:16:45Z","snapshot_observed_at":"2026-08-09T18:26:12.869738Z","submitted_at":"2025-01-28T18:59:44Z","title":"SFT Memorizes, RL Generalizes: A Comparative Study of Foundation Model Post-training","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.17161","snapshot_observed_at":"2026-08-06T18:43:56.867692Z","title":"Le, Sergey Levine, and Yi Ma","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:56.867692Z"},"links":{"cited_paper":"/paper/2501.17161","citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:c5ac835987d172c1f10215fb8e9776ed2d05fd6850105756b5e955a27f680d79","observation_id":"a182342f-0939-4216-a5ad-c4014354a35f","resolution":{"observed_at":"2026-08-06T18:43:56.867692Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-14T02:43:01.480086Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-06T18:43:56.929622Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:56.929622Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:4b02079d1e5a3d85017d29777d3dcf26e5521f379a99e54c40663327af03fff3","observation_id":"be213f86-508b-4eec-bd9d-7ce01221e1b0","resolution":{"observed_at":"2026-08-06T18:43:56.929622Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:43:57.001314Z","title":"Min, Gail E","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:57.001314Z"},"links":{"citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:5e4759392c0843dab84fa78bd10ea1bb5c7a3fad37724d32a7f07cc8f83e3f84","observation_id":"d7be902d-f847-4705-b203-fc78ebf9373e","resolution":{"observed_at":"2026-08-06T18:43:57.001314Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01006","last_updated":"2024-10-31T23:44:31Z","snapshot_observed_at":"2026-08-16T13:46:53.603876Z","submitted_at":"2024-06-03T05:36:57Z","title":"SemCoder: Training Code Language Models with Comprehensive Semantics Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01006","snapshot_observed_at":"2026-08-06T18:43:57.092562Z","title":"Min, Gail Kaiser, Junfeng Yang, and Baishakhi Ray","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:57.092562Z"},"links":{"cited_paper":"/paper/2406.01006","citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:fcd8f6df91564d25eb28effbc693255860be07014a01f276e7b3c2ab2467af77","observation_id":"00ba49be-2db1-48ac-bad9-5f6480f21142","resolution":{"observed_at":"2026-08-06T18:43:57.092562Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:44:01.587568Z","title":"Min, Gail E","venue":null,"work_id":"4c44beec-3e35-40d0-a337-f3f4a57fac00","year":2024},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:57.183728Z"},"links":{"citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:8bff322fbfd018f55c38798128c2787b2ebd8d1883cfebe0007428f8f3267a2b","observation_id":"c2b5160f-9463-4fb9-82bb-23e5fda2d77e","resolution":{"observed_at":"2026-08-06T18:44:01.615383Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:43:57.248831Z","title":"Kaiser, Wei Le, and Baishakhi Ray","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:57.248831Z"},"links":{"citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:5c47b28e909fd9709f177e1846c87db6b3b558460614dc587a4731d977381ab1","observation_id":"63d3f85b-8bcb-49b3-acb4-4b9c2a70d255","resolution":{"observed_at":"2026-08-06T18:43:57.248831Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:43:57.306817Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:57.306817Z"},"links":{"citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:60b9fe9bc4437c97445fdb354f4d58e123c0658ec508804488e2945f50310024","observation_id":"f8999b04-aa9a-4d01-b602-c9f5584ecc5c","resolution":{"observed_at":"2026-08-06T18:43:57.306817Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:44:01.500678Z","title":null,"venue":null,"work_id":"2295b38f-c26f-45e7-96fa-42dad2e0a989","year":2024},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:57.360557Z"},"links":{"citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:7a307eaf474f62f05749ac27a622fc460d825cbb025cd09184a672d474941735","observation_id":"9933a054-1d12-4da6-8fde-87d0afd5cedc","resolution":{"observed_at":"2026-08-06T18:44:01.538429Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04127","last_updated":"2025-01-10T14:31:21Z","snapshot_observed_at":"2026-08-16T13:45:30.496997Z","submitted_at":"2024-06-06T14:49:06Z","title":"Are We Done with MMLU?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04127","snapshot_observed_at":"2026-08-06T18:43:57.410818Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:57.410818Z"},"links":{"cited_paper":"/paper/2406.04127","citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:9e2567dc403d991baf6d45f7510585dbf8ae907e52d476fb9819087f87d0d39a","observation_id":"9ee7c40b-946e-46c1-9991-3dda6ac8a2ae","resolution":{"observed_at":"2026-08-06T18:43:57.410818Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13731","last_updated":"2025-01-23T15:04:22Z","snapshot_observed_at":"2026-08-11T04:10:47.517119Z","submitted_at":"2025-01-23T15:04:22Z","title":"Pseudocode-Injection Magic: Enabling LLMs to Tackle Graph Computational Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13731","snapshot_observed_at":"2026-08-06T18:43:57.462689Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:57.462689Z"},"links":{"cited_paper":"/paper/2501.13731","citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:288f6629513498721bec98150c08f037e7a4f8cd9c0f988a87b772369d85dbce","observation_id":"2524e37f-3834-445b-ad3f-2766ed9f426a","resolution":{"observed_at":"2026-08-06T18:43:57.462689Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:44:01.401944Z","title":null,"venue":null,"work_id":"d2d62463-8328-448e-a4b0-ef4cf9a2e85a","year":2024},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:57.510655Z"},"links":{"citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:07e237355bb28e17a86722d6f2822a7b5c78b8056270af1b76599b65bb4c4ed1","observation_id":"7ea1aca0-17cf-44ce-bab3-d6912a0b2bc1","resolution":{"observed_at":"2026-08-06T18:44:01.449710Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:44:01.319516Z","title":null,"venue":null,"work_id":"14c2308f-b354-4034-8c41-b9daaf2f56ba","year":2024},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:57.558937Z"},"links":{"citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:15fd23b8a61cc212bd8cf86aeee5b7272a26f585dc90fd72dfab0b576066d2b4","observation_id":"89b15af0-e0ae-43b1-876b-86fffe9e5918","resolution":{"observed_at":"2026-08-06T18:44:01.355905Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-08-15T12:33:55.451951Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-06T18:43:57.608469Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:57.608469Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:0ade5d9231ef918c155c26b46c5fb84f385297b1dc1a576b93997e7a69d54016","observation_id":"b770dbb4-18f9-4cd7-8879-0dc42f6fe197","resolution":{"observed_at":"2026-08-06T18:43:57.608469Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.14196","last_updated":"2024-01-26T09:23:11Z","snapshot_observed_at":"2026-08-06T22:40:28.707813Z","submitted_at":"2024-01-25T14:17:53Z","title":"DeepSeek-Coder: When the Large Language Model Meets Programming -- The Rise of Code Intelligence","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.14196","snapshot_observed_at":"2026-08-06T18:43:57.651700Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:57.651700Z"},"links":{"cited_paper":"/paper/2401.14196","citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:5d6ba9aaee89bc6a6e1c98cb77880fe98e85873b4504a6dd3993bdf5af9cb3df","observation_id":"e5066ef0-7bd0-4cd0-84fa-93f0e1316849","resolution":{"observed_at":"2026-08-06T18:43:57.651700Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2103.03874","last_updated":"2021-11-08T21:30:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-03-05T18:59:39Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2103.03874","snapshot_observed_at":"2026-08-06T18:43:57.708756Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:57.708756Z"},"links":{"cited_paper":"/paper/2103.03874","citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:99a3765ec71f8d1b7485a2c4bfca8239bcddf61cbf1b4303002d4e6807433804","observation_id":"81f6f63a-f19c-4e28-9807-46b81b6a8dbd","resolution":{"observed_at":"2026-08-06T18:43:57.708756Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:43:57.785882Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:57.785882Z"},"links":{"citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:d41bccc88a46670ba610f5f779895430911b6c5e0b3ef3bbe15efbdab53541ff","observation_id":"4a38939c-a5c4-4976-952f-08e297d0c85a","resolution":{"observed_at":"2026-08-06T18:43:57.785882Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12186","last_updated":"2024-11-12T13:24:25Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-18T17:57:57Z","title":"Qwen2.5-Coder Technical Report","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12186","snapshot_observed_at":"2026-08-06T18:43:57.839394Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:57.839394Z"},"links":{"cited_paper":"/paper/2409.12186","citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:a7114bf13166b8fc78266946b53ce827916df6887d203bcc32d41b57e95942c6","observation_id":"0f54101d-8a92-4ba9-ae5e-9ccf6a4cad05","resolution":{"observed_at":"2026-08-06T18:43:57.839394Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.16720","last_updated":"2026-04-30T02:46:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-21T18:04:31Z","title":"OpenAI o1 System Card","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.16720","snapshot_observed_at":"2026-08-06T18:43:57.895612Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:57.895612Z"},"links":{"cited_paper":"/paper/2412.16720","citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:e11b36f14ce78e6fb6b35702b4b5070658c68f596e96c94cb8bccd1a1c97f1c4","observation_id":"c01f51ce-a523-47e5-ba9f-cdb2e14be61a","resolution":{"observed_at":"2026-08-06T18:43:57.895612Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.07974","last_updated":"2024-06-06T17:41:21Z","snapshot_observed_at":"2026-08-16T07:05:57.323612Z","submitted_at":"2024-03-12T17:58:04Z","title":"LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.07974","snapshot_observed_at":"2026-08-06T18:43:57.969802Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:57.969802Z"},"links":{"cited_paper":"/paper/2403.07974","citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:64c1d0f20b24ad4a71cce5dd6ded167016313bfdd42b4c9b2db9f7f5c58ef8c4","observation_id":"fc6c8bd9-bc18-447b-8b49-db539a07e775","resolution":{"observed_at":"2026-08-06T18:43:57.969802Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:44:01.239229Z","title":null,"venue":null,"work_id":"9d8cfeb9-b632-4283-bcdd-284a147e1486","year":2025},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:58.020560Z"},"links":{"citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:6baf1b17b6fbb8fbeeab9cfc88be703b283a8cae46481840c9552a822d54d308","observation_id":"f775d378-1c6f-41d0-93b1-f1171f34f13f","resolution":{"observed_at":"2026-08-06T18:44:01.283436Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:43:58.069083Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:58.069083Z"},"links":{"citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:fd833ad1537862e4ee84a26cdda4cfac53429f8d01a16e28d4830b4ca9f836f7","observation_id":"493bf569-5162-4e7c-a312-81eb33bb145a","resolution":{"observed_at":"2026-08-06T18:43:58.069083Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15124","last_updated":"2025-04-14T22:39:09Z","snapshot_observed_at":"2026-08-17T14:11:00.232598Z","submitted_at":"2024-11-22T18:44:04Z","title":"Tulu 3: Pushing Frontiers in Open Language Model Post-Training","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15124","snapshot_observed_at":"2026-08-06T18:43:58.194426Z","title":"Miranda, Alisa Liu, Nouha Dziri, Shane Lyu, Yuling Gu, Saumya Malik, Victoria Graf, Jena D","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:58.194426Z"},"links":{"cited_paper":"/paper/2411.15124","citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:8a629918b733e4525602b11215e8a6c19c1326ddd9d8c153a26de1ad1b202452","observation_id":"5ca038d3-4070-4412-93fe-258d7d9f6090","resolution":{"observed_at":"2026-08-06T18:43:58.194426Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.07316","last_updated":"2025-05-21T13:38:27Z","snapshot_observed_at":"2026-08-16T15:36:15.556886Z","submitted_at":"2025-02-11T07:26:50Z","title":"CodeI/O: Condensing Reasoning Patterns via Code Input-Output Prediction","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.07316","snapshot_observed_at":"2026-08-06T18:43:58.301904Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:58.301904Z"},"links":{"cited_paper":"/paper/2502.07316","citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:96139ff1846851a084ab731a14b9f59e281c92ca7127297948eb7291076528f5","observation_id":"705bca43-4958-4332-8364-62c8169b340e","resolution":{"observed_at":"2026-08-06T18:43:58.301904Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.17123","last_updated":"2026-05-21T12:25:44Z","snapshot_observed_at":"2026-08-14T05:22:26.374179Z","submitted_at":"2025-05-21T17:59:12Z","title":"MTR-Bench: A Comprehensive Benchmark for Multi-Turn Reasoning Evaluation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.17123","snapshot_observed_at":"2026-08-06T18:43:58.430688Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:58.430688Z"},"links":{"cited_paper":"/paper/2505.17123","citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:dfe0d62ca2fc59554d7fb0aa73c29cf78a005d3d4611887792ff600577506e06","observation_id":"c8d37bcf-d2fd-4899-ae03-7805c3e87a0c","resolution":{"observed_at":"2026-08-06T18:43:58.430688Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.11393","last_updated":"2025-05-26T03:52:24Z","snapshot_observed_at":"2026-08-16T12:57:54.209064Z","submitted_at":"2025-02-17T03:24:02Z","title":"HellaSwag-Pro: A Large-Scale Bilingual Benchmark for Evaluating the Robustness of LLMs in Commonsense Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.11393","snapshot_observed_at":"2026-08-06T18:43:58.486504Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:58.486504Z"},"links":{"cited_paper":"/paper/2502.11393","citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:d41a933432b0054b06e5f05e25878eaadce3887332b3ed5da82ace1001cecd27","observation_id":"abe9e047-01dc-4272-a012-e5869a9c0d20","resolution":{"observed_at":"2026-08-06T18:43:58.486504Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:44:01.119115Z","title":null,"venue":null,"work_id":"6455c316-ce40-48a9-9cdd-fcfef9b01d44","year":2024},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:58.628898Z"},"links":{"citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:1695fa77904325801a67559da7e52e5242618ebbe4d74877378c7f441cb7c964","observation_id":"cf41a461-6ba7-4a9d-97a6-6c755177b0fc","resolution":{"observed_at":"2026-08-06T18:44:01.158488Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01100","last_updated":"2025-07-15T01:14:25Z","snapshot_observed_at":"2026-08-15T18:38:02.815401Z","submitted_at":"2025-02-03T06:44:49Z","title":"ZebraLogic: On the Scaling Limits of LLMs for Logical Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.01100","snapshot_observed_at":"2026-08-06T18:43:58.771660Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:58.771660Z"},"links":{"cited_paper":"/paper/2502.01100","citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:d55906022d702fcf9a758ab611fc33f9fe2ead4b9fc563ee574a86ad9d205781","observation_id":"351480f6-b042-44dc-a133-69877045730e","resolution":{"observed_at":"2026-08-06T18:43:58.771660Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15382","last_updated":"2025-03-30T23:56:09Z","snapshot_observed_at":"2026-08-17T20:27:35.522440Z","submitted_at":"2024-11-22T23:54:37Z","title":"On the Impact of Fine-Tuning on Chain-of-Thought Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15382","snapshot_observed_at":"2026-08-06T18:43:58.869826Z","title":"Lobo, Chirag Agarwal, and Himabindu Lakkaraju","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:58.869826Z"},"links":{"cited_paper":"/paper/2411.15382","citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:22540733461934d4147d59c3d6b3d387c87203f177f342c39337ba98c0d294ae","observation_id":"711345b0-5fcf-4d58-aae5-ae6f48db94f9","resolution":{"observed_at":"2026-08-06T18:43:58.869826Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:43:58.929656Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:58.929656Z"},"links":{"citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:e57ecc838a02fa568e2e08083663eb8e22e57d5a350ae9d35c11d9e471ea911f","observation_id":"43a09bcd-2a34-4758-b151-876c4c53b183","resolution":{"observed_at":"2026-08-06T18:43:58.929656Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06526","last_updated":"2025-03-01T12:34:10Z","snapshot_observed_at":"2026-08-16T13:11:16.770516Z","submitted_at":"2024-10-09T03:56:50Z","title":"KOR-Bench: Benchmarking Language Models on Knowledge-Orthogonal Reasoning Tasks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.06526","snapshot_observed_at":"2026-08-06T18:43:58.987141Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:58.987141Z"},"links":{"cited_paper":"/paper/2410.06526","citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:cd6f4150fa3ed363bd69ef7ab29a1432936ab8b3edba870be9d471bc9d1890f2","observation_id":"b02a2f70-fb09-44ee-a3ad-ec2234bab7ff","resolution":{"observed_at":"2026-08-06T18:43:58.987141Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-08-17T18:50:07.059564Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-06T18:43:59.046822Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:59.046822Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:5eada69fc61d34eeb8aa54edb75506aad311465c81adf29ca338eaca15e5c0c3","observation_id":"0270f618-f1c0-4595-84dd-2b347d0ec3fc","resolution":{"observed_at":"2026-08-06T18:43:59.046822Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:43:59.109983Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:59.109983Z"},"links":{"citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:1b542b9a940a626e7a959c7392ed6015ed48ca37d069710e1bd6ad36ccf31225","observation_id":"17feeeea-e0d2-4784-99ce-83f14916907e","resolution":{"observed_at":"2026-08-06T18:43:59.109983Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:43:59.155801Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:59.155801Z"},"links":{"citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:068979bf81027fa1b2bf44f03db33b450d5c570c352c8300155f74d9143c8851","observation_id":"67fb2b71-9929-4c31-8644-0cc18dac0ee4","resolution":{"observed_at":"2026-08-06T18:43:59.155801Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-06T18:43:59.205671Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:59.205671Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:fa9b8c5b94d6f4b6fcd10d894c20c300e1b085d05130849610e55726b1945d56","observation_id":"2b48e226-7534-4af7-a9c1-1b5fcc0ea1a2","resolution":{"observed_at":"2026-08-06T18:43:59.205671Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:44:01.013179Z","title":null,"venue":null,"work_id":"d29ce8f6-6278-4d96-b6a0-68652ff09255","year":2024},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:59.267078Z"},"links":{"citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:c633e81b0d08bc8be1195b28fbc82f19fca3d5f0f7742f1710f4aba64bf9a6be","observation_id":"62d1e1e1-8df0-4206-9df7-212aa93f435d","resolution":{"observed_at":"2026-08-06T18:44:01.049327Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:43:59.311348Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:59.311348Z"},"links":{"citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:39386d06ac736df9ce3b49e2acf35c3e6fe6631c38cbd1d176bc96765f37265d","observation_id":"4c774638-4500-4e1d-beac-2ba61ae4e4b4","resolution":{"observed_at":"2026-08-06T18:43:59.311348Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14739","last_updated":"2025-03-28T15:21:44Z","snapshot_observed_at":"2026-08-15T17:40:05.031382Z","submitted_at":"2025-02-20T17:05:58Z","title":"SuperGPQA: Scaling LLM Evaluation across 285 Graduate Disciplines","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14739","snapshot_observed_at":"2026-08-06T18:43:59.388088Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:59.388088Z"},"links":{"cited_paper":"/paper/2502.14739","citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:0c5da1a1f13d2d353c7ee4613bfff41f501d691512cc48fa5a697452d05d3a94","observation_id":"d7dde5ff-939b-4562-b0e6-95af6b34ff72","resolution":{"observed_at":"2026-08-06T18:43:59.388088Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.10176","last_updated":"2024-11-03T03:48:02Z","snapshot_observed_at":"2026-08-16T17:24:49.433939Z","submitted_at":"2024-02-15T18:26:11Z","title":"OpenMathInstruct-1: A 1.8 Million Math Instruction Tuning Dataset","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.10176","snapshot_observed_at":"2026-08-06T18:43:59.471326Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:59.471326Z"},"links":{"cited_paper":"/paper/2402.10176","citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:57b2ccf88523080d8809e9d6c52a0c6d9bb4c24840d08bd871efa60add497688","observation_id":"ff4a753f-d84a-4aaf-b0b8-e6b537bc62dc","resolution":{"observed_at":"2026-08-06T18:43:59.471326Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:44:00.936251Z","title":null,"venue":null,"work_id":"5e3881b4-2005-49dd-b2ac-d58ab26c6cbc","year":2025},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:59.551743Z"},"links":{"citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:f167142060b7222dd2d90951839c9b785e33cf312224691b67ce1342a279e85b","observation_id":"88366e24-9723-48b8-933d-b688da49350d","resolution":{"observed_at":"2026-08-06T18:44:00.969329Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03731","last_updated":"2023-10-05T17:52:09Z","snapshot_observed_at":"2026-08-16T14:54:42.760719Z","submitted_at":"2023-10-05T17:52:09Z","title":"MathCoder: Seamless Code Integration in LLMs for Enhanced Mathematical Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03731","snapshot_observed_at":"2026-08-06T18:43:59.614488Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:59.614488Z"},"links":{"cited_paper":"/paper/2310.03731","citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:9039d6a8dd665e0ba94bd022d2ad3dc07eb482f7d723498a015bdfe191c23445","observation_id":"38559dbf-f1ae-4979-aae0-abcdcdc53f6f","resolution":{"observed_at":"2026-08-06T18:43:59.614488Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:44:00.846024Z","title":"Le, Ed H","venue":null,"work_id":"e26c5bf2-e228-4be9-bebe-5cac8bcc227c","year":2023},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:59.669231Z"},"links":{"citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:973f16b2ba5ec499f7e2df0fa152296d3a3880b6eec692931db4e7f65505f18a","observation_id":"db14ff2f-6586-4fba-a676-ab1b1c522760","resolution":{"observed_at":"2026-08-06T18:44:00.892980Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:44:00.744514Z","title":null,"venue":null,"work_id":"0c3b9d38-1eb9-4e12-9ed8-22850de848ad","year":2024},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:59.719102Z"},"links":{"citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:7275a573535a10ab08a8d72e7fcc9aba237f4234d62f919c6b5d5219485627bd","observation_id":"2e0b8cd6-af3b-40a5-8efa-256c0de4201b","resolution":{"observed_at":"2026-08-06T18:44:00.785136Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:44:00.668984Z","title":"Chi, Tatsunori Hashimoto, Oriol Vinyals, Percy Liang, Jeff Dean, and William Fedus","venue":null,"work_id":"e19a6eea-1ffa-4749-88ce-6ddb683c006b","year":2022},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:59.764041Z"},"links":{"citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:152a43814ef394c71feda523181230b8e3b002dd33aa80782933b354a5d62bf4","observation_id":"bf2044ff-eb66-4004-bf51-d582acf36600","resolution":{"observed_at":"2026-08-06T18:44:00.703241Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:44:00.538391Z","title":null,"venue":null,"work_id":"bce85b87-33c4-47f8-9cc1-864f9d65d164","year":2022},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:59.808781Z"},"links":{"citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:c0edd413ab3493938577e782a5c5699c0d2ff7d2d3c24af230ce293a4e91010d","observation_id":"92874828-cc70-4d5e-b29b-aec626613d50","resolution":{"observed_at":"2026-08-06T18:44:00.584735Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.14655","last_updated":"2025-04-20T15:28:16Z","snapshot_observed_at":"2026-08-16T11:41:37.083470Z","submitted_at":"2025-04-20T15:28:16Z","title":"LeetCodeDataset: A Temporal Dataset for Robust Evaluation and Efficient Training of Code LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.14655","snapshot_observed_at":"2026-08-06T18:43:59.848050Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:59.848050Z"},"links":{"cited_paper":"/paper/2504.14655","citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:465f0c78081ca7e253cd68d9b04314bf00d80feb6fe390830adad6d61304434a","observation_id":"b8b0e9da-da06-47ad-8c4d-09724995ea1a","resolution":{"observed_at":"2026-08-06T18:43:59.848050Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12122","last_updated":"2024-09-18T16:45:37Z","snapshot_observed_at":"2026-08-17T18:51:13.219936Z","submitted_at":"2024-09-18T16:45:37Z","title":"Qwen2.5-Math Technical Report: Toward Mathematical Expert Model via Self-Improvement","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12122","snapshot_observed_at":"2026-08-06T18:43:59.893251Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:59.893251Z"},"links":{"cited_paper":"/paper/2409.12122","citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:6b0389f28201bb9b70a363847c24617344ef63a4f1131fcfb3145ed34a83197a","observation_id":"99a139d5-32ad-498a-880e-a8739508a026","resolution":{"observed_at":"2026-08-06T18:43:59.893251Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.01825","last_updated":"2023-09-13T03:57:29Z","snapshot_observed_at":"2026-08-16T18:17:11.146813Z","submitted_at":"2023-08-03T15:34:01Z","title":"Scaling Relationship on Learning Mathematical Reasoning with Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.01825","snapshot_observed_at":"2026-08-06T18:43:59.932862Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:59.932862Z"},"links":{"cited_paper":"/paper/2308.01825","citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:0c78638640c0d9cfcf788fba87fb46e219f5ea793384b1970f50c7f3b1e398c6","observation_id":"498cf9e9-8b5a-42f6-9848-f9b4f31190ec","resolution":{"observed_at":"2026-08-06T18:43:59.932862Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.13837","last_updated":"2025-11-24T06:11:04Z","snapshot_observed_at":"2026-08-14T14:03:15.178702Z","submitted_at":"2025-04-18T17:59:56Z","title":"Does Reinforcement Learning Really Incentivize Reasoning Capacity in LLMs Beyond the Base Model?","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.13837","snapshot_observed_at":"2026-08-06T18:43:59.959396Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:59.959396Z"},"links":{"cited_paper":"/paper/2504.13837","citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:b2d180e3fc2a29b586dbfb508fdd7c4ca761ca7e4c6cc49e3f06d3186a94ef67","observation_id":"536563bd-f0ae-4a38-ac31-fe7cc09dc943","resolution":{"observed_at":"2026-08-06T18:43:59.959396Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1410.4615","last_updated":"2015-02-19T15:33:35Z","snapshot_observed_at":"2026-08-14T23:15:21.933877Z","submitted_at":"2014-10-17T01:35:12Z","title":"Learning to Execute","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1410.4615","snapshot_observed_at":"2026-08-06T18:43:59.965594Z","title":null,"venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:59.965594Z"},"links":{"cited_paper":"/paper/1410.4615","citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:18acf190936ed5ff64fb008355d615b9f4c1b3d577126e535fe652667174da68","observation_id":"85f4a8c5-380c-40c7-ac76-6aedd63ed9e2","resolution":{"observed_at":"2026-08-06T18:43:59.965594Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:43:59.970269Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:59.970269Z"},"links":{"citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:8b8bd514a7877d213476baf6f9931c0206723039f07716e74ccb3e993794de90","observation_id":"f7351096-4071-4a91-8b50-e1129681d1ba","resolution":{"observed_at":"2026-08-06T18:43:59.970269Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:43:59.987191Z","title":"online\" 'onlinestring :=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:59.987191Z"},"links":{"citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:638d9f721689feb0b537a507c5135ae38d1fa1cfa63c98c50a68b5524a07ecc4","observation_id":"a5e5406b-ede2-49b5-b534-889a281da0ac","resolution":{"observed_at":"2026-08-06T18:43:59.987191Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:44:00.020826Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code","version":2},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-06T18:44:00.020826Z"},"links":{"citing_paper":"/paper/2507.07498"},"observation_digest":"sha256:f0b8d78f913eaf78b5f6c458685748fac3cadd054bc7760c3301a2f5f62d9cc0","observation_id":"9cff075c-7ed1-42d4-929a-396445ae561e","resolution":{"observed_at":"2026-08-06T18:44:00.020826Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2507.07498","last_updated":"2025-07-14T07:10:51Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-15T12:03:04.307569Z","submitted_at":"2025-07-10T07:34:05Z","title":"Teaching LLM to Reason: Reinforcement Learning from Algorithmic Problems without Code"},"reference_resolution":{"displayed":56,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":53,"verified_exact":0,"verified_fuzzy":3},"total_outbound_references":56},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 56 of 56 outbound references and 1 inbound Pith citation observation for arXiv:2507.07498."}