{"as_of":"2026-08-10T09:51:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:64d9c0c67e100b99ac43eb6c2ef49ac2186f4e325ec45f4e43980898c85a52a9","coverage":[{"denominator":87,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":87,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T13:05:31.372865Z","state":"measured"},{"denominator":88,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":88,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-16T20:25:35.573805Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-16T20:28:24.120617Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"cited_work":{"arxiv_id":"2505.22756","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.22756","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"John Schulman, Filip Wolski, Prafulla Dhariwal, Alec Radford, and Oleg Klimov","venue":null,"work_id":"c84694fb-d5cf-448e-bf98-7fe642a18f66","year":null},"citing_paper":{"arxiv_id":"2512.18857","last_updated":"2026-05-07T11:50:40Z","snapshot_observed_at":"2026-07-06T22:39:43.359149Z","submitted_at":"2025-12-21T19:01:35Z","title":"CORE: Concept-Oriented Reinforcement for Bridging the Definition-Application Gap in Mathematical Reasoning","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-16T20:25:35.573805Z"},"links":{"cited_paper":"/paper/2505.22756","citing_paper":"/paper/2512.18857"},"observation_digest":"sha256:836814bcd596b062be1f4c8914352a254fc44dfc2c6c0027d73658622a29e1fa","observation_id":"603b071b-5bbf-48a0-b0f7-b90891d6b21a","resolution":{"observed_at":"2026-05-16T20:28:24.122726Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2505.22756/citation-record","integrity":"/paper/2505.22756/integrity","json":"/paper/2505.22756/citation-record.json","paper":"/paper/2505.22756"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-07T13:05:23.720759Z","title":"DeepSeekMath: Pushing the limits of mathematical reasoning in open language models.arXiv [cs.CL], 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:23.720759Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:d768cfaeb37d4ca081602bf0e1bc2e90fd681f30bed7f58f276c1062c3e2bbec","observation_id":"82283523-60c5-48bb-83ab-9c3d27fa1cb5","resolution":{"observed_at":"2026-08-07T13:05:23.720759Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1037/0003-066x.63.4.215","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"High stakes testing in higher education and employment: appraising the evidence for validity and fairness.Am","venue":"American Psychologist","work_id":"d53fce5a-ed1a-48e9-8cac-ef6396dad762","year":2008},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:23.852152Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:22d1afa3a9906bcc8e3b8d3080fb0a46d92ccdfa1611384999fd6354cffdc7cf","observation_id":"7ca46755-9a23-481a-bc2a-8c1a994a0634","resolution":{"observed_at":"2026-08-07T13:05:32.024612Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.3102/0013189x07306523","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"High-stakes testing and curricular control: A qualitative metasynthesis.Educ","venue":"Educational Researcher","work_id":"fc73cbb6-0003-4689-9acb-76f34cf067f9","year":2007},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:23.953514Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:8ed1bafa86ea31067cf3e5ebf1eb20fa1c9520cc6c711e830a306d01cc5df5b9","observation_id":"8c4ac61e-2af2-4da5-8914-d17b1e46dcf4","resolution":{"observed_at":"2026-08-07T13:05:31.818803Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1016/s0160-2896(97)90014-3","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"Why g matters: The complexity of everyday life.Intelligence, 24(1): 79–132, 1997","venue":"Intelligence","work_id":"ffdaeaab-f99a-4a60-bf91-a23e107f44bd","year":1997},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:24.093228Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:afc04828010317cd13e72357ced55138bced54f68951ec10e85483c6b3ddff44","observation_id":"50d718c8-8fb3-4e90-93f2-fa56358bff9e","resolution":{"observed_at":"2026-08-07T13:05:31.632620Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:24.239878Z","title":"Homo heuristicus: Why biased minds make better inferences","venue":null,"work_id":null,"year":2011},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:24.239878Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:a11ea7e8d8f974681dc545eb5be4f3dfb92d8c9e737f34628bc707cb75878e78","observation_id":"d2693040-663b-4906-9c9a-a3529bdc9050","resolution":{"observed_at":"2026-08-07T13:05:24.239878Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:24.391541Z","title":"Farrar, Straus and Giroux, 2011","venue":null,"work_id":null,"year":2011},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:24.391541Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:dcfc32f8e906b80f6adc987b589cb25703221903b2a1692e2800f9cbdcc91419","observation_id":"2f7643f8-a36d-40e1-b608-8aa98d11cb6f","resolution":{"observed_at":"2026-08-07T13:05:24.391541Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:24.548005Z","title":null,"venue":null,"work_id":null,"year":1964},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:24.548005Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:c4b846e23cc3503fcd2d85f092797e0511b42dd4d401a9f93198998f4b431523","observation_id":"3d8e41cd-b126-4c4f-ad5c-9470fcffa4a4","resolution":{"observed_at":"2026-08-07T13:05:24.548005Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:24.683120Z","title":"MAWPS: A math word problem repository","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:24.683120Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:ca0e3cc4128477e957aad177355dd5d80fd7e064d80ad4061936379f70cc4538","observation_id":"12408603-e58e-4c7a-86fe-ff0e36aa2e65","resolution":{"observed_at":"2026-08-07T13:05:24.683120Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:24.792439Z","title":"Deep neural solver for math word problems","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:24.792439Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:0eb1c9ed05e95e7dff4d63f0cb57eaaa4ebccc697599be909ac719e6ceea3aef","observation_id":"bf9916bc-be26-4b77-9516-8e310d2c54d3","resolution":{"observed_at":"2026-08-07T13:05:24.792439Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1905.13319","last_updated":"2019-05-30T21:28:12Z","snapshot_observed_at":"2026-07-06T07:56:54.143277Z","submitted_at":"2019-05-30T21:28:12Z","title":"MathQA: Towards Interpretable Math Word Problem Solving with Operation-Based Formalisms","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1905.13319","snapshot_observed_at":"2026-08-07T13:05:24.942656Z","title":"MathQA: Towards interpretable math word problem solving with operation-based formalisms.arXiv [cs.CL], 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:24.942656Z"},"links":{"cited_paper":"/paper/1905.13319","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:1cf28c9f833cb82e86118955f526f50504e12d4790f77e191d3255a72c0bd2a7","observation_id":"d47cd48e-6fa8-4937-ba91-2cd6bc16f89f","resolution":{"observed_at":"2026-08-07T13:05:24.942656Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-07T01:45:38.840969Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-07T13:05:25.060454Z","title":"Training verifiers to solve math word problems.arXiv [cs.LG], 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:25.060454Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:a6b124058be6938e1e1441a99833425a1cd059253f6f106ea4fdd6487ed974d9","observation_id":"395b9c2e-6189-4518-9789-013b3f37bb10","resolution":{"observed_at":"2026-08-07T13:05:25.060454Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2103.03874","last_updated":"2021-11-08T21:30:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-03-05T18:59:39Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2103.03874","snapshot_observed_at":"2026-08-07T13:05:25.168303Z","title":"Measuring mathematical problem solving with the MATH dataset","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:25.168303Z"},"links":{"cited_paper":"/paper/2103.03874","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:54eae151bcdef3adc1797746da1b185d07ec1ec8a7bd08a8db9d5789397bfe9d","observation_id":"476212b4-4c9f-44ab-9ef4-d1aca498bf8f","resolution":{"observed_at":"2026-08-07T13:05:25.168303Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:25.229947Z","title":"A survey of deep learning for mathematical reasoning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:25.229947Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:9e7a2fed11a18caf9c24b734c5de7eb23e5474bd310636c19bcb0844de085cbf","observation_id":"78cc8178-93b7-4f8e-b868-a3ee1fbbff6d","resolution":{"observed_at":"2026-08-07T13:05:25.229947Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:25.304597Z","title":"OlympiadBench: A challenging benchmark for promoting AGI with olympiad-level bilingual multimodal scientific problems.arXiv [cs.CL], 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:25.304597Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:a8637a2aa61caa560bf57401011b27a504d2dec11b3970683d02c9ea91d14e14","observation_id":"80f19518-4bf7-485b-80b4-91ebbf3beff0","resolution":{"observed_at":"2026-08-07T13:05:25.304597Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.11214","last_updated":"2024-11-03T17:14:37Z","snapshot_observed_at":"2026-08-08T18:18:53.692980Z","submitted_at":"2024-07-15T19:57:15Z","title":"PutnamBench: Evaluating Neural Theorem-Provers on the Putnam Mathematical Competition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.11214","snapshot_observed_at":"2026-08-07T13:05:25.371971Z","title":"PutnamBench: Evaluating neural theorem-provers on the putnam mathematical competition.arXiv [cs.AI], 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:25.371971Z"},"links":{"cited_paper":"/paper/2407.11214","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:7787da9def289763bcbf144fdddc3fc23ad5635b599261e6a2e3e879a82b6373","observation_id":"023917f0-443d-47fe-aa5f-43cee0c92f78","resolution":{"observed_at":"2026-08-07T13:05:25.371971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07985","last_updated":"2024-12-24T04:04:30Z","snapshot_observed_at":"2026-08-10T07:44:25.135572Z","submitted_at":"2024-10-10T14:39:33Z","title":"Omni-MATH: A Universal Olympiad Level Mathematic Benchmark For Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07985","snapshot_observed_at":"2026-08-07T13:05:25.427253Z","title":"Omni-MATH: A universal olympiad level mathematic benchmark for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:25.427253Z"},"links":{"cited_paper":"/paper/2410.07985","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:933ac8585b811319d40ca71863d837347bcaf6f2b4c07e569b9bf655a72dad76","observation_id":"69415067-5e5a-4773-98af-3df7ece59e28","resolution":{"observed_at":"2026-08-07T13:05:25.427253Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2109.00110","last_updated":"2022-02-28T06:03:23Z","snapshot_observed_at":"2026-08-09T06:26:57.793642Z","submitted_at":"2021-08-31T23:21:12Z","title":"MiniF2F: a cross-system benchmark for formal Olympiad-level mathematics","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.00110","snapshot_observed_at":"2026-08-07T13:05:25.496990Z","title":"MiniF2F: a cross-system benchmark for formal olympiad-level mathematics.arXiv [cs.AI], 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:25.496990Z"},"links":{"cited_paper":"/paper/2109.00110","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:fcfc5d46b96013686b9770a361ce77362d77455732488c606f8bfbd29391cda6","observation_id":"7f690810-a145-4ea6-ae50-4024a039bc09","resolution":{"observed_at":"2026-08-07T13:05:25.496990Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.01112","last_updated":"2021-06-07T21:58:06Z","snapshot_observed_at":"2026-08-03T23:23:04.398852Z","submitted_at":"2021-03-24T03:14:48Z","title":"NaturalProofs: Mathematical Theorem Proving in Natural Language","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.01112","snapshot_observed_at":"2026-08-07T13:05:25.544101Z","title":"NaturalProofs: Mathematical theorem proving in natural language.arXiv [cs.IR], 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:25.544101Z"},"links":{"cited_paper":"/paper/2104.01112","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:8080ae7658fd7b00b69e3043c846abadeb9ea2a7e41b2481cef7d9e7b672c716","observation_id":"1b80a8eb-dc37-4a35-b1f4-6f6b0ba24090","resolution":{"observed_at":"2026-08-07T13:05:25.544101Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2209.09513","last_updated":"2022-10-17T07:46:47Z","snapshot_observed_at":"2026-08-02T03:10:19.073297Z","submitted_at":"2022-09-20T07:04:24Z","title":"Learn to Explain: Multimodal Reasoning via Thought Chains for Science Question Answering","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2209.09513","snapshot_observed_at":"2026-08-07T13:05:25.637408Z","title":"Learn to explain: Multimodal reasoning via thought chains for science question answering.arXiv [cs.CL], 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:25.637408Z"},"links":{"cited_paper":"/paper/2209.09513","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:78b19c0a0254738aecf1ba2a5734fd427e03ba65a646f7aca8a40414a178feda","observation_id":"e76fe449-c3c7-4302-b8c7-9d3b6bdb3e5b","resolution":{"observed_at":"2026-08-07T13:05:25.637408Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2105.14517","last_updated":"2022-01-11T03:50:31Z","snapshot_observed_at":"2026-07-06T11:14:04.454155Z","submitted_at":"2021-05-30T12:34:17Z","title":"GeoQA: A Geometric Question Answering Benchmark Towards Multimodal Numerical Reasoning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2105.14517","snapshot_observed_at":"2026-08-07T13:05:25.722333Z","title":"GeoQA: A geometric question answering benchmark towards multimodal numerical reasoning","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:25.722333Z"},"links":{"cited_paper":"/paper/2105.14517","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:920e1552ae8baae0740760b18e35869117ad6b78ef050d85a1cd7418262a7195","observation_id":"eebbef18-5422-4da1-bb9e-31a4d2dd1f0f","resolution":{"observed_at":"2026-08-07T13:05:25.722333Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:25.793569Z","title":"MMMU: A massive multi-discipline multimodal understanding and reasoning benchmark for expert AGI.arXiv [cs.CL], 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:25.793569Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:c0a6d4fdb5c8726e1fc172ca9cebca74f1e5d98d5d19e97de03796026cb559a8","observation_id":"408b344b-8672-4e41-9b9f-da920bdb606a","resolution":{"observed_at":"2026-08-07T13:05:25.793569Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.02155","last_updated":"2022-03-04T07:04:42Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-03-04T07:04:42Z","title":"Training language models to follow instructions with human feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.02155","snapshot_observed_at":"2026-08-07T13:05:25.929835Z","title":"Training language models to follow instructions with human feedback.arXiv [cs.CL], 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:25.929835Z"},"links":{"cited_paper":"/paper/2203.02155","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:a17007cdf8b589ae395ce8445c061426b41332b305901adbdc13f135eb72571b","observation_id":"3d0b7976-580a-4ba0-930d-8798bc2fb459","resolution":{"observed_at":"2026-08-07T13:05:25.929835Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.14465","last_updated":"2022-05-20T13:52:54Z","snapshot_observed_at":"2026-08-05T03:08:42.211484Z","submitted_at":"2022-03-28T03:12:15Z","title":"STaR: Bootstrapping Reasoning With Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.14465","snapshot_observed_at":"2026-08-07T13:05:26.000236Z","title":"STaR: Bootstrapping reasoning with reasoning.arXiv [cs.LG], 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:26.000236Z"},"links":{"cited_paper":"/paper/2203.14465","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:139894a6e8901d7713ec9e9391ac7ea05e22054a0d437a0e8a9d8d2f297ef256","observation_id":"1f525c57-5dd2-417b-ad07-854da3f25b12","resolution":{"observed_at":"2026-08-07T13:05:26.000236Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.04751","last_updated":"2023-10-30T20:36:20Z","snapshot_observed_at":"2026-08-04T08:05:09.732049Z","submitted_at":"2023-06-07T19:59:23Z","title":"How Far Can Camels Go? Exploring the State of Instruction Tuning on Open Resources","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.04751","snapshot_observed_at":"2026-08-07T13:05:26.072736Z","title":"How far can camels go? exploring the state of instruction tuning on open resources.arXiv [cs.CL], 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:26.072736Z"},"links":{"cited_paper":"/paper/2306.04751","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:edf7e743f4fded13c7910c3152821829db9c5665772a35cbabf115b197231215","observation_id":"9b3e43d9-4873-4318-871d-78a4c845558a","resolution":{"observed_at":"2026-08-07T13:05:26.072736Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.08998","last_updated":"2023-08-21T10:23:42Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-17T14:12:48Z","title":"Reinforced Self-Training (ReST) for Language Modeling","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.08998","snapshot_observed_at":"2026-08-07T13:05:26.221091Z","title":"Reinforced self-training (ReST) for language modeling.arXiv [cs.CL], 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:26.221091Z"},"links":{"cited_paper":"/paper/2308.08998","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:4a5c4db39dec91f3756ded035bbaad81b719faaf479318650308ecd4010a3ff2","observation_id":"77752e58-f40c-434e-8b2d-e35c14acdc89","resolution":{"observed_at":"2026-08-07T13:05:26.221091Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:26.276783Z","title":"Proximal policy optimization algorithms.arXiv [cs.LG], 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:26.276783Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:5dde52350b3538568ce8a783ecbebdd67b6ece02aa7263ebe30b7bab831d28db","observation_id":"8135a8b8-b4ff-4950-b067-d76f37834b19","resolution":{"observed_at":"2026-08-07T13:05:26.276783Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:26.374522Z","title":"VinePPO: Unlocking RL potential for LLM reasoning through refined credit assignment.arXiv [cs.LG], 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:26.374522Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:c411cbe9e2e36932c21c9f0f89a45a8b79d9ef0acfab574609a8f81a8793ed6f","observation_id":"90cfacd7-9eba-465d-8b0d-48b1b0e38a8e","resolution":{"observed_at":"2026-08-07T13:05:26.374522Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.05118","last_updated":"2025-04-11T02:54:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-07T14:21:11Z","title":"VAPO: Efficient and Reliable Reinforcement Learning for Advanced Reasoning Tasks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.05118","snapshot_observed_at":"2026-08-07T13:05:26.479274Z","title":"V APO: Efficient and reliable reinforcement learning for advanced reasoning tasks.arXiv [cs.AI], 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:26.479274Z"},"links":{"cited_paper":"/paper/2504.05118","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:0095a048999384a58111469efe09aa6d09389401e685c3ad7284552e8202c8cd","observation_id":"0f4fdb04-2f99-40c7-9417-1ec1686aa36a","resolution":{"observed_at":"2026-08-07T13:05:26.479274Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15124","last_updated":"2025-04-14T22:39:09Z","snapshot_observed_at":"2026-08-08T12:58:42.430328Z","submitted_at":"2024-11-22T18:44:04Z","title":"Tulu 3: Pushing Frontiers in Open Language Model Post-Training","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15124","snapshot_observed_at":"2026-08-07T13:05:26.568756Z","title":"TÜLU 3: Pushing frontiers in open language model post-training.arXiv [cs.CL], 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:26.568756Z"},"links":{"cited_paper":"/paper/2411.15124","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:8c73df5666de3a615624d80418c2b4863f0f1e214806a5230e6e44ccf8f336a4","observation_id":"7631f79c-4c16-4afd-9064-b59b4fa72769","resolution":{"observed_at":"2026-08-07T13:05:26.568756Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12599","last_updated":"2025-06-03T02:14:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T02:48:14Z","title":"Kimi k1.5: Scaling Reinforcement Learning with LLMs","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12599","snapshot_observed_at":"2026-08-07T13:05:26.666052Z","title":"Kimi k1.5: Scaling reinforcement learning with LLMs.arXiv [cs.AI], 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:26.666052Z"},"links":{"cited_paper":"/paper/2501.12599","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:5b997ea106e5213fbd6fe0c9a4a733644f53b7de6d431442579459893973cecd","observation_id":"6b00a84d-b48c-4a25-8dc3-f2f1356c0bc5","resolution":{"observed_at":"2026-08-07T13:05:26.666052Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-07T13:05:26.751890Z","title":"DeepSeek-R1: Incentivizing reasoning capability in LLMs via reinforcement learning.arXiv [cs.CL], 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:26.751890Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:a3d90a385d4742aa8d54726d4a24f4ba3069758b6bdd6abc6c4ead01bf772771","observation_id":"3be659cc-c28b-4bd2-98b5-ce03edbe2da5","resolution":{"observed_at":"2026-08-07T13:05:26.751890Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.18892","last_updated":"2025-08-06T08:42:32Z","snapshot_observed_at":"2026-07-06T20:57:57.039376Z","submitted_at":"2025-03-24T17:06:10Z","title":"SimpleRL-Zoo: Investigating and Taming Zero Reinforcement Learning for Open Base Models in the Wild","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.18892","snapshot_observed_at":"2026-08-07T13:05:26.857948Z","title":"SimpleRL-zoo: Investigating and taming zero reinforcement learning for open base models in the wild, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:26.857948Z"},"links":{"cited_paper":"/paper/2503.18892","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:a15a4ef59d22cd3bc8b71726128c030a2ca79f50cb1fcdcda28856657e218b32","observation_id":"28d3b52c-838f-43a9-b9ee-2a894563b6e2","resolution":{"observed_at":"2026-08-07T13:05:26.857948Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:26.921792Z","title":"Light-R1: Curriculum SFT, DPO and RL for long COT from scratch and beyond.arXiv [cs.CL],","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:26.921792Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:af3cab2d01427854312bbb6e8e403c2ad134dc869ca200440b45476f095ca63e","observation_id":"5861a4df-ac1f-4cec-a1e8-44ab4561a788","resolution":{"observed_at":"2026-08-07T13:05:26.921792Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:36.768427Z","title":"Let’s verify step by step.arXiv [cs.LG],","venue":null,"work_id":"a18d6b23-9dd3-4d47-8cc7-2a0bbe76aa3b","year":null},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:27.095654Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:6b4f54598273fb443951ec8704228db7f3b5573be2eed70b3ff15e7c32e1056a","observation_id":"30aeb803-7bae-44c8-ac49-ea7e4be49c1e","resolution":{"observed_at":"2026-08-07T13:05:36.816620Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.16720","last_updated":"2026-04-30T02:46:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-21T18:04:31Z","title":"OpenAI o1 System Card","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.16720","snapshot_observed_at":"2026-08-07T13:05:27.300656Z","title":"OpenAI o1 system card.arXiv [cs.AI], 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:27.300656Z"},"links":{"cited_paper":"/paper/2412.16720","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:d8c9a258096775cd5bb0de5124d0ccc0b1bd41dd9eb5fb2744c43016657ecdf5","observation_id":"c404877b-310a-4657-8ab3-3b80ad51c2d9","resolution":{"observed_at":"2026-08-07T13:05:27.300656Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.20783","last_updated":"2025-10-06T09:30:03Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-26T17:59:14Z","title":"Understanding R1-Zero-Like Training: A Critical Perspective","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.20783","snapshot_observed_at":"2026-08-07T13:05:27.416461Z","title":"Understanding R1-zero-like training: A critical perspective.arXiv [cs.LG], 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:27.416461Z"},"links":{"cited_paper":"/paper/2503.20783","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:956ed268af34c2e0b60fc31048b4b84adcf3069dcf65b32ffe08ff577fe36dd3","observation_id":"9d7a3efa-87d4-4fe9-b964-45bc6be1b8eb","resolution":{"observed_at":"2026-08-07T13:05:27.416461Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:36.662327Z","title":"DeepCoder: A fully open-source 14B coder at O3-mini level","venue":null,"work_id":"9fb54339-bf9c-4a89-a90d-5f6b5f41d3f8","year":2025},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:27.467393Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:1e5ef6718e78020d7e35b1b246ed193009063b55198f9e9a12b50681009aa03a","observation_id":"8cb34916-823f-4fdd-9062-80cfe37000e0","resolution":{"observed_at":"2026-08-07T13:05:36.723108Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.07615","last_updated":"2025-04-14T15:15:54Z","snapshot_observed_at":"2026-08-08T20:11:45.308315Z","submitted_at":"2025-04-10T10:05:15Z","title":"VLM-R1: A Stable and Generalizable R1-style Large Vision-Language Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.07615","snapshot_observed_at":"2026-08-07T13:05:27.610508Z","title":"VLM-R1: A stable and generalizable R1-style large vision-language model.arXiv [cs.CV], 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:27.610508Z"},"links":{"cited_paper":"/paper/2504.07615","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:677d387cdbc881d3320cde40809415fd01906ecd84f745eb62ef3502864a2484","observation_id":"31fd81e0-c229-41c3-95dc-513d334cf53f","resolution":{"observed_at":"2026-08-07T13:05:27.610508Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:27.668035Z","title":"GRPO-LEAD: A difficulty-aware reinforcement learning approach for concise mathematical reasoning in language models.arXiv [cs.CL], 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:27.668035Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:61ebf4be1484b43f4d82c6ad4a4c1e8b70c640ea20dbb0a36dce54060c490252","observation_id":"c03b72ad-4cbb-40a5-8879-5924e6e2d775","resolution":{"observed_at":"2026-08-07T13:05:27.668035Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.12216","last_updated":"2025-06-03T17:02:25Z","snapshot_observed_at":"2026-08-07T16:01:44.170215Z","submitted_at":"2025-04-16T16:08:45Z","title":"d1: Scaling Reasoning in Diffusion Large Language Models via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.12216","snapshot_observed_at":"2026-08-07T13:05:27.735188Z","title":"D1: Scaling reasoning in diffusion large language models via reinforcement learning.arXiv [cs.CL], 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:27.735188Z"},"links":{"cited_paper":"/paper/2504.12216","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:50f6df38fd1eeffc3a2be270c16a9772c5343b657f37c1145902fc8f4164533e","observation_id":"c273ed96-ece8-4163-bc0c-c5b15126e428","resolution":{"observed_at":"2026-08-07T13:05:27.735188Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.13837","last_updated":"2025-11-24T06:11:04Z","snapshot_observed_at":"2026-07-06T21:11:34.701779Z","submitted_at":"2025-04-18T17:59:56Z","title":"Does Reinforcement Learning Really Incentivize Reasoning Capacity in LLMs Beyond the Base Model?","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.13837","snapshot_observed_at":"2026-08-07T13:05:27.917867Z","title":"Does reinforcement learning really incentivize reasoning capacity in LLMs beyond the base model?arXiv [cs.AI], 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:27.917867Z"},"links":{"cited_paper":"/paper/2504.13837","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:0a1f112ef63c6df36bb7f850063a87402f2c868db61253e47172895c68f6efc3","observation_id":"c85ad0a3-3c96-4073-8eef-7a3c9af437bb","resolution":{"observed_at":"2026-08-07T13:05:27.917867Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:36.576436Z","title":"Assessing diversity collapse in reasoning","venue":null,"work_id":"8ccd99f6-dfe5-42d0-b3d2-36640ff93f94","year":null},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:28.017100Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:e79322c988a5ddd56859c943f1d55105b2da924162e780dd9bff4ee4d0e745fb","observation_id":"202eeab6-03a5-45c2-b3f0-84071ba8a5b6","resolution":{"observed_at":"2026-08-07T13:05:36.606466Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2009.01325","last_updated":"2022-02-15T19:09:36Z","snapshot_observed_at":"2026-08-06T19:18:51.065158Z","submitted_at":"2020-09-02T19:54:41Z","title":"Learning to summarize from human feedback","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.01325","snapshot_observed_at":"2026-08-07T13:05:28.179005Z","title":"Learning to summarize from human feedback","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:28.179005Z"},"links":{"cited_paper":"/paper/2009.01325","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:38c5d03e0203afe42330591366b1ee9d85feaaa9327784613f35033d868f89a9","observation_id":"caad0a5c-5339-47b2-8771-3ffa6e252669","resolution":{"observed_at":"2026-08-07T13:05:28.179005Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.11171","last_updated":"2023-03-07T17:57:37Z","snapshot_observed_at":"2026-07-06T12:50:22.773056Z","submitted_at":"2022-03-21T17:48:52Z","title":"Self-Consistency Improves Chain of Thought Reasoning in Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.11171","snapshot_observed_at":"2026-08-07T13:05:28.252859Z","title":"Self-consistency improves chain of thought reasoning in language models.arXiv [cs.CL], 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:28.252859Z"},"links":{"cited_paper":"/paper/2203.11171","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:5fdb979be3b4831ad2c9571cad716431b4f3b9b244eee94b3104080369723b79","observation_id":"300928c1-f5db-49e9-bc3d-4da23f35e5e4","resolution":{"observed_at":"2026-08-07T13:05:28.252859Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:36.422201Z","title":null,"venue":null,"work_id":"2793227e-59e4-4d9f-87c9-03760b742b40","year":null},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:28.080765Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:be1dabd7a68dee09d8bb71d2c5a0831d4bb89f212ed11d4477123ea7ac277dd1","observation_id":"377a623d-8e95-45b4-98d6-7ee65bb17484","resolution":{"observed_at":"2026-08-07T13:05:36.490896Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:36.290856Z","title":"HybridFlow: A flexible and efficient RLHF framework","venue":null,"work_id":"ca6325e4-0905-491b-a47e-b5cdc2e0d78c","year":2024},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:28.404599Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:c98d3152da5e31eb5aefff2133868e346149ce15e0772953598ca381e944c2d9","observation_id":"9b7d2b68-a4b2-42c5-9e02-e9b943bc5ffa","resolution":{"observed_at":"2026-08-07T13:05:36.352870Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21787","last_updated":"2024-12-30T19:03:24Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:57:25Z","title":"Large Language Monkeys: Scaling Inference Compute with Repeated Sampling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21787","snapshot_observed_at":"2026-08-07T13:05:28.475062Z","title":"Large language monkeys: Scaling inference compute with repeated sampling, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:28.475062Z"},"links":{"cited_paper":"/paper/2407.21787","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:b049f334a9ebec6343914329ffe0138c9facb52ac30848b68bb7fdfab55d6b7d","observation_id":"8c32ea10-34a5-4d17-b422-a0c7c4f6cf21","resolution":{"observed_at":"2026-08-07T13:05:28.475062Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:28.346576Z","title":"Qwen2.5: A party of foundation models, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:28.346576Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:c4fb304746bc22dc258846d5cb18b269400d299d6e0135b46cddd8b9221ec80a","observation_id":"b38a8d1c-2b8c-4739-bb75-a0519f87d6e7","resolution":{"observed_at":"2026-08-07T13:05:28.346576Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.10963","last_updated":"2024-06-25T03:14:10Z","snapshot_observed_at":"2026-08-05T17:51:38.668510Z","submitted_at":"2024-02-13T20:16:29Z","title":"GLoRe: When, Where, and How to Improve LLM Reasoning via Global and Local Refinements","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.10963","snapshot_observed_at":"2026-08-07T13:05:28.672657Z","title":"GLoRe: When, where, and how to im- prove LLM reasoning via global and local refinements.arXiv [cs.CL], 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:28.672657Z"},"links":{"cited_paper":"/paper/2402.10963","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:c6b2215b017fa210b9a29452a79cf3654e106e1d47d9b6329bee95ee5e03c715","observation_id":"58f50a08-3af2-4c3f-a900-dd231a7ac5dd","resolution":{"observed_at":"2026-08-07T13:05:28.672657Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01307","last_updated":"2025-08-15T15:21:46Z","snapshot_observed_at":"2026-08-08T21:52:29.510852Z","submitted_at":"2025-03-03T08:46:22Z","title":"Cognitive Behaviors that Enable Self-Improving Reasoners, or, Four Habits of Highly Effective STaRs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01307","snapshot_observed_at":"2026-08-07T13:05:28.746323Z","title":"Cognitive behaviors that enable self-improving reasoners, or, four habits of highly effective STaRs.arXiv [cs.CL], 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:28.746323Z"},"links":{"cited_paper":"/paper/2503.01307","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:b68988ce0609690d56a811a53471320e83b336abceb1cfafa67e5684e7b64473","observation_id":"0cb832ef-4dc8-4fc8-893c-185ec6aa5fac","resolution":{"observed_at":"2026-08-07T13:05:28.746323Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:28.588850Z","title":"Math-shepherd: Verify and reinforce LLMs step-by-step without human annotations","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:28.588850Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:cc871c6c173802a4ddb890ec521aee33164537d2e2491104b4c5373b70a363e9","observation_id":"f2d66ea0-bd66-4731-8131-24dbbd7792a2","resolution":{"observed_at":"2026-08-07T13:05:28.588850Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.16410","last_updated":"2023-10-25T06:49:26Z","snapshot_observed_at":"2026-07-06T16:38:13.945197Z","submitted_at":"2023-10-25T06:49:26Z","title":"Bridging the Human-AI Knowledge Gap: Concept Discovery and Transfer in AlphaZero","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.16410","snapshot_observed_at":"2026-08-07T13:05:28.914635Z","title":"Bridging the human-ai knowledge gap: Concept discovery and transfer in alphazero, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:28.914635Z"},"links":{"cited_paper":"/paper/2310.16410","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:bb43634994f53e6ec90ced8e049befa432b6725e1de25cbc9675b150fa04694c","observation_id":"5671997c-a6f8-4cb7-b8e4-8341b5fe5c51","resolution":{"observed_at":"2026-08-07T13:05:28.914635Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:28.958067Z","title":"To backtrack or not to backtrack: When sequential search limits model reasoning, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:28.958067Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:05e0e0c9eada80b300480aa69b35160a6ffc35ae3ae81da13c4ce264c296faef","observation_id":"fb99a8b4-a71a-4647-8511-87319c0ece22","resolution":{"observed_at":"2026-08-07T13:05:28.958067Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01815","last_updated":"2017-12-05T18:45:38Z","snapshot_observed_at":"2026-08-02T00:39:04.960144Z","submitted_at":"2017-12-05T18:45:38Z","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01815","snapshot_observed_at":"2026-08-07T13:05:28.844213Z","title":"Mastering chess and shogi by self-play with a general reinforcement learning algorithm, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:28.844213Z"},"links":{"cited_paper":"/paper/1712.01815","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:d5eb7f4286b2f081d4424755c9249ddeaa9b011c36d5518e04baeab2380c416a","observation_id":"f9703d2b-f199-4dd6-8544-44b9b91e86e5","resolution":{"observed_at":"2026-08-07T13:05:28.844213Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.19370","last_updated":"2024-12-11T07:53:57Z","snapshot_observed_at":"2026-08-01T15:02:18.622098Z","submitted_at":"2024-06-27T17:50:05Z","title":"Emergence of Hidden Capabilities: Exploring Learning Dynamics in Concept Space","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.19370","snapshot_observed_at":"2026-08-07T13:05:29.106731Z","title":"Emergence of hidden capabilities: Exploring learning dynamics in concept space, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:29.106731Z"},"links":{"cited_paper":"/paper/2406.19370","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:e9388730fddffdca48fe83df6942870346199d707cb4972528026d71d30378eb","observation_id":"00921f7a-d3e7-4539-9be6-d111fdb705ff","resolution":{"observed_at":"2026-08-07T13:05:29.106731Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.01003","last_updated":"2025-05-02T05:25:53Z","snapshot_observed_at":"2026-08-09T04:43:46.597431Z","submitted_at":"2024-12-01T23:35:53Z","title":"Competition Dynamics Shape Algorithmic Phases of In-Context Learning","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.01003","snapshot_observed_at":"2026-08-07T13:05:29.166990Z","title":"Competition dynamics shape algorithmic phases of in-context learning, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:29.166990Z"},"links":{"cited_paper":"/paper/2412.01003","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:c65cf27b8f76fdf72cdd1564012191cbdb795445fb679e889bbbbf8611616654","observation_id":"983f2b90-2eb1-4251-8468-5e795a89176f","resolution":{"observed_at":"2026-08-07T13:05:29.166990Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.09336","last_updated":"2025-07-25T21:42:43Z","snapshot_observed_at":"2026-08-07T04:10:49.606943Z","submitted_at":"2023-10-13T18:00:59Z","title":"Compositional Abilities Emerge Multiplicatively: Exploring Diffusion Models on a Synthetic Task","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.09336","snapshot_observed_at":"2026-08-07T13:05:29.034561Z","title":"Dick, and Hidenori Tanaka","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:29.034561Z"},"links":{"cited_paper":"/paper/2310.09336","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:b5f573b681af8a6861026ab930f87365926deffacc3e0a9744d67e19763a0de3","observation_id":"40c252c9-7fd7-4054-9778-f7f4a678cee6","resolution":{"observed_at":"2026-08-07T13:05:29.034561Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12578","last_updated":"2024-09-07T20:10:28Z","snapshot_observed_at":"2026-07-06T19:04:43.716629Z","submitted_at":"2024-08-22T17:44:22Z","title":"A Percolation Model of Emergence: Analyzing Transformers Trained on a Formal Language","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12578","snapshot_observed_at":"2026-08-07T13:05:29.293988Z","title":"Dick, and Hidenori Tanaka","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:29.293988Z"},"links":{"cited_paper":"/paper/2408.12578","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:063cd38b414ad553e8b2c63238f88a768dad10eb356d5fa152113c231ad84fd1","observation_id":"3466d3f9-47d8-47f2-8dfb-6605a6ebaeea","resolution":{"observed_at":"2026-08-07T13:05:29.293988Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10264","last_updated":"2024-08-21T15:12:37Z","snapshot_observed_at":"2026-08-06T20:21:15.838210Z","submitted_at":"2024-07-14T16:12:57Z","title":"What Makes and Breaks Safety Fine-tuning? A Mechanistic Study","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.10264","snapshot_observed_at":"2026-08-07T13:05:29.382934Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:29.382934Z"},"links":{"cited_paper":"/paper/2407.10264","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:5dffad06d7704a8c177c3a3b5bd4a2fb033c671a8e6ebed251be4d01a995d9d7","observation_id":"e7173816-d650-4d15-9788-34446761e211","resolution":{"observed_at":"2026-08-07T13:05:29.382934Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.14402","last_updated":"2024-07-16T10:33:12Z","snapshot_observed_at":"2026-08-07T23:54:23.509475Z","submitted_at":"2023-09-25T17:50:41Z","title":"Physics of Language Models: Part 3.2, Knowledge Manipulation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.14402","snapshot_observed_at":"2026-08-07T13:05:29.251296Z","title":"Physics of language models: Part 3.2, knowledge manipula- tion, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:29.251296Z"},"links":{"cited_paper":"/paper/2309.14402","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:0d14ae74149470a077ef984c24e3309965235ca59a3916ef733954e6b39bd9ca","observation_id":"73b5f83a-1b8a-421e-b2e9-8867fbde1ed8","resolution":{"observed_at":"2026-08-07T13:05:29.251296Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20571","last_updated":"2025-10-24T10:02:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-29T09:24:30Z","title":"Reinforcement Learning for Reasoning in Large Language Models with One Training Example","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20571","snapshot_observed_at":"2026-08-07T13:05:29.522894Z","title":"Reinforcement learning for reasoning in large language models with one training example, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:29.522894Z"},"links":{"cited_paper":"/paper/2504.20571","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:f858cb127c3fb83c87fd2fb2166fce25b9d99be4470ecf29d6c8664ceec9e975","observation_id":"939d6d91-0d16-4a1e-b604-84f2348605d9","resolution":{"observed_at":"2026-08-07T13:05:29.522894Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1706.03762","last_updated":"2023-08-02T00:41:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-06-12T17:57:34Z","title":"Attention Is All You Need","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1706.03762","snapshot_observed_at":"2026-08-07T13:05:29.607118Z","title":"Gomez, Lukasz Kaiser, and Illia Polosukhin","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:29.607118Z"},"links":{"cited_paper":"/paper/1706.03762","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:e311a25f248469b9a85d1af1d0afbe435a266b036b077b581109c9393692077b","observation_id":"020e1ca4-07b6-4e29-a3c9-03ef1fae2248","resolution":{"observed_at":"2026-08-07T13:05:29.607118Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.13506","last_updated":"2024-01-13T23:51:39Z","snapshot_observed_at":"2026-08-07T23:56:01.694697Z","submitted_at":"2023-03-23T17:58:43Z","title":"The Quantization Model of Neural Scaling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.13506","snapshot_observed_at":"2026-08-07T13:05:29.453050Z","title":"Michaud, Ziming Liu, Uzay Girit, and Max Tegmark","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:29.453050Z"},"links":{"cited_paper":"/paper/2303.13506","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:a40ac4e0368aa7f689d098e761a8df11876f84657b3f32fc1b7a796a5566f235","observation_id":"f0241cbe-04b2-4060-a40e-9de13a9bac7c","resolution":{"observed_at":"2026-08-07T13:05:29.453050Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:36.016792Z","title":"sharpens","venue":null,"work_id":"bea75383-bc2c-4c36-9292-7f6510fb95bf","year":2020},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:29.748740Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:daf0cdd83cc8829f86f25e992b780fd63bd1d4fba5fb9ab49f0953befd59647c","observation_id":"7a2ec3ed-c4b4-4270-beb7-61ffa2e7e411","resolution":{"observed_at":"2026-08-07T13:05:36.070631Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:36.161135Z","title":"Qwen2 technical report.arXiv preprint arXiv:2407","venue":null,"work_id":"df605b2e-ef59-4de2-9043-bbbdb570429d","year":2024},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:29.654648Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:3e7b9bda7f2415782c94b1b665a20a89e9f8a0e9be6a71f5be2143ac3e896b13","observation_id":"abce898d-46c9-456e-9e80-8bcbab57d3d2","resolution":{"observed_at":"2026-08-07T13:05:36.195705Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:35.874348Z","title":"the first two decimal places are 0.27","venue":null,"work_id":"adf41d1e-9338-4743-821b-245f5b64fba4","year":null},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:29.822515Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:a60c5152d33df201c2179bb2ab7f2a215d84afd5c6a3cbf226761ef11eccba5e","observation_id":"bc13b8ce-dfe3-4359-971e-f549cc7ccd6e","resolution":{"observed_at":"2026-08-07T13:05:35.950753Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:35.748596Z","title":null,"venue":null,"work_id":"504ff28d-7913-4a32-96bc-a1494561686f","year":null},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:29.905371Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:bf06b45eaa0bcde1c623b47c4a050eb63f710e3720aa74fff386dbc3af771549","observation_id":"1bf8d851-4e77-4906-bc28-583a4a2688c2","resolution":{"observed_at":"2026-08-07T13:05:35.806060Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:35.610554Z","title":null,"venue":null,"work_id":"142d6b3f-fbe8-4134-b0ef-cb2f78fcd927","year":null},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:29.979846Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:557324aebf1b1de5312d4b1c12af38410227dc19ca2b21ce9389027ce74af36a","observation_id":"463d0080-27a5-4a15-aa0d-ee1a0297ffe3","resolution":{"observed_at":"2026-08-07T13:05:35.663631Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:35.481118Z","title":null,"venue":null,"work_id":"113af579-8958-4220-bdfb-30bfb517ef14","year":null},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:30.073137Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:ca165bd0db0a860228c0aad26f26eff482b13c6e23edf273dc126d6380f0cccc","observation_id":"4f49bd8b-049d-4271-bd94-c41f4d0ad79a","resolution":{"observed_at":"2026-08-07T13:05:35.550297Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:35.388998Z","title":null,"venue":null,"work_id":"3805c38f-b80b-49cc-ace8-f1bd0b108a50","year":null},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:30.147910Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:fddbc8e52171603c565348da6b636e7340cb3d7755a06e86c098de273e8170cc","observation_id":"9b26f74e-b9ec-4fc6-a922-822de6ba836c","resolution":{"observed_at":"2026-08-07T13:05:35.436789Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:35.270569Z","title":null,"venue":null,"work_id":"a72d37bd-f7f2-4c9a-9530-7943fa9e2473","year":null},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:30.205663Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:adf73140d9526a79f22c5971afa3f5c0954b9087d753ff2dc8494a626a5a7e4a","observation_id":"7a5eeabd-3e7a-419f-802d-0d0f86afdfb1","resolution":{"observed_at":"2026-08-07T13:05:35.316602Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:35.123922Z","title":"Since 2 is prime but not odd, it should not be in the intersection","venue":null,"work_id":"f477bbfd-5d3e-433e-8f2e-9d26690da929","year":null},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:30.278206Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:ed125800f9d4d68a74daafe82e3ffb21625785305b25c4b343501bd0437bbedb","observation_id":"85bda93f-464b-4e0b-aae0-46e1a616e1b3","resolution":{"observed_at":"2026-08-07T13:05:35.203697Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:35.016953Z","title":null,"venue":null,"work_id":"b4e45cee-ddc3-4a8d-8a4f-8591760747c1","year":null},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:30.379472Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:bd89af4c3d2b6239e3aa903d37b657a7edd26521e7af0a9ef783b17bef33787d","observation_id":"b375b355-d8c4-4609-ba17-b32800fad553","resolution":{"observed_at":"2026-08-07T13:05:35.058424Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:34.860450Z","title":null,"venue":null,"work_id":"c7ced9c0-14d5-4642-b708-2a0cafc6a0ea","year":null},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:30.454881Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:2ce2919c1befcfd1033523f7ff5ddbe01bf11ff3dff87a7328bf5997e46d8739","observation_id":"50950dab-abfe-4c32-bcef-2b78f61dedf2","resolution":{"observed_at":"2026-08-07T13:05:34.910887Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:34.699844Z","title":null,"venue":null,"work_id":"18277d4d-4a15-406d-a716-730771863831","year":null},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:30.541458Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:9a8c2c5d345804bb0aa5fe417e106fb03611b8716a20b0e28eb307298430e795","observation_id":"9b5f1754-28e7-480f-981e-cb4275866ea6","resolution":{"observed_at":"2026-08-07T13:05:34.779509Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:34.591489Z","title":null,"venue":null,"work_id":"9b6a4855-d7aa-4c93-a114-0b9d5dfc94db","year":null},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:30.601536Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:762941b6e4309add59940abd874da514d66116382e42b28b551a537fb809dc13","observation_id":"7c59d434-5112-42fa-80b2-004de786e120","resolution":{"observed_at":"2026-08-07T13:05:34.648944Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:34.436595Z","title":"2 is odd","venue":null,"work_id":"e2e194fe-dd5c-488e-8a1d-f2949fb907b7","year":2010},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:30.691035Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:ce6906e85b0dca64544a66a0dae5ee0270c79cc5431e34317f65d6aaaa89ec07","observation_id":"42a8f567-11f5-41ae-b027-0a60bc6c599f","resolution":{"observed_at":"2026-08-07T13:05:34.489359Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:34.277946Z","title":"unknown\" token <unk>, beginning of sequence token <bos>, padding token <pad>, prompt/completion separation token “:","venue":null,"work_id":"fcef4279-b1c2-454d-8340-1ef7058cbd74","year":null},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:30.793829Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:71a95609d3d78d2bf855f70f46142c4fd5791daf4b54f13b51066640d465279e","observation_id":"716ce123-e3d0-4770-b3eb-effe5cca9a19","resolution":{"observed_at":"2026-08-07T13:05:34.339866Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:34.090790Z","title":"An entry TS(s, a)specifies the next state s′ resulting from taking action a in state s","venue":null,"work_id":"e03363fc-8c63-4f72-b37f-15027fe2d49a","year":null},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:30.880373Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:16ed577b2eebf1ac4eb86fb59b1515cfbec08f9b67f6e0cfcba7b3c7a467dcd9","observation_id":"48a50a48-5b8a-4b6b-84d2-200dbc55b067","resolution":{"observed_at":"2026-08-07T13:05:34.191214Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:33.921113Z","title":"An entry TP (p, a)specifies the next problem state p′ when action a is taken while the current problem state is p","venue":null,"work_id":"0f3d30cf-e626-4bf4-b651-3746a435c53a","year":null},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:30.952180Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:eeaec8fec597497fa15a067f2de227750b4ba866f31c63c0d9f5ee0bf1136bcd","observation_id":"f8594844-f8ad-4e67-b0f1-77ca35bae6f2","resolution":{"observed_at":"2026-08-07T13:05:34.015931Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:33.752679Z","title":"Each row s contains the probability distribution Pα(A|S=s) over actions, generated from Dirichlet(αtrain ·1 NA )","venue":null,"work_id":"00bc0fe9-56f9-4cff-bf73-a384b3ae69ba","year":null},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:31.042538Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:f7774f4843e25eb37d0ce71daec7268d36d9a2211690dcb8ee7fcc070d4a16ae","observation_id":"d02c6e70-38ea-4fb1-a57c-7afd717fb9c5","resolution":{"observed_at":"2026-08-07T13:05:33.814601Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:33.577570Z","title":"Each row a contains the probability distribution Pβ(C|A=a) over contexts, generated from Dirichlet(βtrain ·1 NC )","venue":null,"work_id":"d9c5d100-01fd-4d78-ab23-162393c4e4d2","year":null},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:31.116321Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:6b21801b565856e229e70b17c68ee838518bad53fd5133609ad20f32865a9b4e","observation_id":"649e6a37-5108-4eee-a444-2cb213a730a6","resolution":{"observed_at":"2026-08-07T13:05:33.647234Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:33.410173Z","title":"question","venue":null,"work_id":"515676e1-1faf-4309-b516-f6e1110dfd64","year":null},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:31.224853Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:0905744c61b84cad051a11caeca04f276c21190cd932e78ff7a0dc91ef3c5223","observation_id":"f36687f8-d05d-4015-9271-141b9089fdbd","resolution":{"observed_at":"2026-08-07T13:05:33.473480Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:33.296215Z","title":"This k defines the length of the action sequence and consequently the number of state transitions","venue":null,"work_id":"8ac17167-eb7b-459b-8e0e-00539c3fb359","year":null},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:31.305142Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:688aa78d55db86a505a2c4834de31242a6b54def9589e2f33f2b8d26a8ce431f","observation_id":"ff1244d0-4c2b-47ad-8561-1e256a6e36a9","resolution":{"observed_at":"2026-08-07T13:05:33.354446Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:05:33.157731Z","title":"S <s 0>\" Then, for each step j from 0 to k: Append problem state:","venue":null,"work_id":"99738b2e-6917-47a5-8e6f-fe06bff3c087","year":null},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:31.372865Z"},"links":{"citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:4f1eda926c2b27317c452e31af2d619d2f15e0939c0a01e35aa20756f5fe2115","observation_id":"2538a8e8-0e48-4356-8aba-0e826ef65f02","resolution":{"observed_at":"2026-08-07T13:05:33.216756Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.20050","last_updated":"2023-05-31T17:24:00Z","snapshot_observed_at":"2026-08-05T13:11:04.104454Z","submitted_at":"2023-05-31T17:24:00Z","title":"Let's Verify Step by Step","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.20050","snapshot_observed_at":"2026-08-07T13:05:27.240873Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:27.240873Z"},"links":{"cited_paper":"/paper/2305.20050","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:72d49e4286d943546ba287378a90586cb1f71b5d9a011e3f684c8a5a30c75bb1","observation_id":"1dc966e7-705c-4868-ba37-3ba77d1bd988","resolution":{"observed_at":"2026-08-07T13:05:27.240873Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-08T01:11:14.602172Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-07T13:05:27.038237Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:27.038237Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:8a56e3341ab4d105237e2e98e19acbc7eea7a357085a02de8b71599399378947","observation_id":"6d56d6f1-1820-4fc9-865c-f168dcb42c1c","resolution":{"observed_at":"2026-08-07T13:05:27.038237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-10T01:26:26.516912Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?"},"reference_resolution":{"displayed":87,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":67,"verified_exact":3,"verified_fuzzy":17},"total_outbound_references":87},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 87 of 87 outbound references and 1 inbound Pith citation observation for arXiv:2505.22756."}