{"as_of":"2026-08-17T20:27:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:c4cb0d5caef6ec9fe47e5d9f1fbe629ab076005b0cdc1c08c5f7609c4d87b462","coverage":[{"denominator":39,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":39,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:41:42.861182Z","state":"measured"},{"denominator":40,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":40,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-17T06:30:58.91139+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T19:02:41.773793Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-05T19:03:11.044453Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"cited_work":{"arxiv_id":"2505.17988","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.17988","snapshot_observed_at":"2026-08-05T19:03:11.044453Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","venue":"cs.LG","work_id":"369dbedc-4444-47e8-8410-dee6551c5a5d","year":2025},"citing_paper":{"arxiv_id":"2508.13579","last_updated":"2025-08-19T07:24:48Z","snapshot_observed_at":"2026-08-16T04:15:20.275101Z","submitted_at":"2025-08-19T07:24:48Z","title":"Toward Better EHR Reasoning in LLMs: Reinforcement Learning with Expert Attention Guidance","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-05T19:02:41.773793Z"},"links":{"cited_paper":"/paper/2505.17988","citing_paper":"/paper/2508.13579"},"observation_digest":"sha256:60c928ccbb21a17a667979ea052195b5ffc1644690f79b338bfc98f840c39fbd","observation_id":"d1d75131-f636-457c-aea4-1ff0062c884a","resolution":{"observed_at":"2026-08-05T19:03:11.133072Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2505.17988/citation-record","integrity":"/paper/2505.17988/integrity","json":"/paper/2505.17988/citation-record.json","paper":"/paper/2505.17988"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:41:38.977890Z","title":"On exact computation with an infinitely wide neural net","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:38.977890Z"},"links":{"citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:b8bc7a7a426e49f96956136a39df7a86aa86eda91a55b594a0bd1fdde9ab5357","observation_id":"ac8f7d77-a107-4077-89fc-af3e6c7306dc","resolution":{"observed_at":"2026-08-07T14:41:38.977890Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:41:44.895554Z","title":"A general theoretical paradigm to understand learning from human preferences","venue":null,"work_id":"84ec90af-54cb-4b5d-bdcf-6f6f4a694cad","year":2024},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:39.029884Z"},"links":{"citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:68d82c44fac3f0e4c5f93387c01c1ce680d6911ae2301cb5c51acc940984eae9","observation_id":"297d458b-dbcc-4f72-ab49-ff4512919862","resolution":{"observed_at":"2026-08-07T14:41:44.971192Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:41:44.614960Z","title":"Quantifying memorization across neural language models","venue":null,"work_id":"bf75702e-8ada-4c56-bc43-7a9d993b4491","year":2022},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:39.164577Z"},"links":{"citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:4998750f5008d7a2eb66bd1e6bbc2c570b9604c2ebd943c4e6edb88727e558d2","observation_id":"31fed710-f59b-407c-a173-eda5d9b6604a","resolution":{"observed_at":"2026-08-07T14:41:44.730875Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.04318","last_updated":"2025-02-26T17:51:31Z","snapshot_observed_at":"2026-08-16T03:28:56.763206Z","submitted_at":"2024-12-05T16:34:20Z","title":"The Hyperfitting Phenomenon: Sharpening and Stabilizing LLMs for Open-Ended Text Generation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.04318","snapshot_observed_at":"2026-08-07T14:41:39.281110Z","title":"The Hyperfitting Phenomenon : Sharpening and Stabilizing LLMs for Open - Ended Text Generation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:39.281110Z"},"links":{"cited_paper":"/paper/2412.04318","citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:ab1541771dee4373ac3b8e5be0222ea265aa9e111f977155ebe422d7ee6856ed","observation_id":"cf4a11ce-1b22-4ae5-9bd5-12572d62f4d8","resolution":{"observed_at":"2026-08-07T14:41:39.281110Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:41:44.486497Z","title":"Generative AI for Math : Abel , 2023","venue":null,"work_id":"13546b09-5193-41aa-8c0b-fa9c3cb7cb7c","year":2023},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:39.412297Z"},"links":{"citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:00976f41e20b6fb6b5e807667bb5a68f0c676aca885a03631a51e02675c60666","observation_id":"23683e5a-5c50-443f-9523-3803e6c70954","resolution":{"observed_at":"2026-08-07T14:41:44.581085Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17161","last_updated":"2025-05-26T17:16:45Z","snapshot_observed_at":"2026-08-09T18:26:12.869738Z","submitted_at":"2025-01-28T18:59:44Z","title":"SFT Memorizes, RL Generalizes: A Comparative Study of Foundation Model Post-training","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.17161","snapshot_observed_at":"2026-08-07T14:41:39.559892Z","title":"Sft memorizes, rl generalizes: A comparative study of foundation model post-training","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:39.559892Z"},"links":{"cited_paper":"/paper/2501.17161","citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:57b251298b2ce960ad7c0df137d7b34c422b1969578f59d8b00a26db056600c3","observation_id":"52973908-ca4a-4c27-81c0-1b03f73d6470","resolution":{"observed_at":"2026-08-07T14:41:39.559892Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01306","last_updated":"2024-11-19T18:12:45Z","snapshot_observed_at":"2026-08-17T15:29:47.883677Z","submitted_at":"2024-02-02T10:53:36Z","title":"KTO: Model Alignment as Prospect Theoretic Optimization","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01306","snapshot_observed_at":"2026-08-07T14:41:39.679644Z","title":"Kto: Model alignment as prospect theoretic optimization","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:39.679644Z"},"links":{"cited_paper":"/paper/2402.01306","citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:edcc2bbbfaa336d35fd1c5a5a8448a6e6b7a61ec899781734bb20596598fe206","observation_id":"7956933a-7f09-42e8-9f54-64b27e12fbf0","resolution":{"observed_at":"2026-08-07T14:41:39.679644Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:41:44.349703Z","title":"Open R1 : A fully open reproduction of DeepSeek - R1 , January 2025","venue":null,"work_id":"6f3eaa40-1068-4846-bbf8-da36323bc1f4","year":2025},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:39.817035Z"},"links":{"citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:8c403b6c670ba33088befd1e5c9e3f4bfa96123c65e75985ef0d25d9bc3cf0cf","observation_id":"2ad3645d-df46-470d-85af-b60d0ad396a4","resolution":{"observed_at":"2026-08-07T14:41:44.410444Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01307","last_updated":"2025-08-15T15:21:46Z","snapshot_observed_at":"2026-08-14T13:46:28.086398Z","submitted_at":"2025-03-03T08:46:22Z","title":"Cognitive Behaviors that Enable Self-Improving Reasoners, or, Four Habits of Highly Effective STaRs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01307","snapshot_observed_at":"2026-08-07T14:41:39.932382Z","title":"Cognitive behaviors that enable self-improving reasoners, or, four habits of highly effective stars","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:39.932382Z"},"links":{"cited_paper":"/paper/2503.01307","citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:a3bda1de452f621680489d569f12bf3da889a1b6c3fd35ead4e3d4f4cf5e0254","observation_id":"790e28f7-f863-4365-9a47-094c6fe7d020","resolution":{"observed_at":"2026-08-07T14:41:39.932382Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-07T14:41:40.069115Z","title":"The llama 3 herd of models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:40.069115Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:e987a403ab744bfd2a2e60ffd1a40431209684889ac04ee55156eb97d0f425b0","observation_id":"2ec1ae1b-3119-44a3-a895-9b83d959e8e2","resolution":{"observed_at":"2026-08-07T14:41:40.069115Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-08-15T12:33:55.451951Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-07T14:41:40.220148Z","title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:40.220148Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:01ce12d378c843349e21d75daa24a28f8192844bffe31ce097e209084d666c68","observation_id":"fc27d130-dbf0-48e8-8537-4ea25f18070a","resolution":{"observed_at":"2026-08-07T14:41:40.220148Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18362","last_updated":"2023-10-24T14:25:53Z","snapshot_observed_at":"2026-08-16T14:49:19.852945Z","submitted_at":"2023-10-24T14:25:53Z","title":"SoK: Memorization in General-Purpose Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.18362","snapshot_observed_at":"2026-08-07T14:41:40.334702Z","title":"SoK : Memorization in General - Purpose Large Language Models , 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:40.334702Z"},"links":{"cited_paper":"/paper/2310.18362","citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:2ecf637937b4cb731ab693925f76eda832ab46e4214b02ee3675d86cd15d369c","observation_id":"e2e60055-69b5-49b5-91f2-8f32525c57dd","resolution":{"observed_at":"2026-08-07T14:41:40.334702Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2103.03874","last_updated":"2021-11-08T21:30:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-03-05T18:59:39Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2103.03874","snapshot_observed_at":"2026-08-07T14:41:40.442788Z","title":"Measuring mathematical problem solving with the math dataset","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:40.442788Z"},"links":{"cited_paper":"/paper/2103.03874","citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:26adb2c3a3065b370097092f9af7c490c5d0c4c9b2c1fa168040982dda2f3e08","observation_id":"ff073668-a698-4c69-968f-456875c472b6","resolution":{"observed_at":"2026-08-07T14:41:40.442788Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.03262","last_updated":"2025-11-10T15:11:13Z","snapshot_observed_at":"2026-08-16T04:57:30.418363Z","submitted_at":"2025-01-04T02:08:06Z","title":"REINFORCE++: Stabilizing Critic-Free Policy Optimization with Global Advantage Normalization","version":9},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.03262","snapshot_observed_at":"2026-08-07T14:41:40.517731Z","title":"Reinforce++: A simple and efficient approach for aligning large language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:40.517731Z"},"links":{"cited_paper":"/paper/2501.03262","citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:d8b7b88383f5cef68d4edb9870a00bef29f5f5dd72d7fa0119d8abaa4bba750c","observation_id":"9805a4cc-3efa-4317-96f9-3e0f371be348","resolution":{"observed_at":"2026-08-07T14:41:40.517731Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:41:44.225662Z","title":"Neural tangent kernel: Convergence and generalization in neural networks","venue":null,"work_id":"60c54a9e-d373-4db0-9bdb-9d31eae7988b","year":2018},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:40.654506Z"},"links":{"citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:b89d63d1bf08cd29198242aa627e2edcf4a0bd8dc498480d1c3bcd5a3a8aa906","observation_id":"1544743b-b843-42da-ba7f-096799465de8","resolution":{"observed_at":"2026-08-07T14:41:44.283754Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.00856","last_updated":"2024-06-05T08:15:12Z","snapshot_observed_at":"2026-08-17T14:37:36.721385Z","submitted_at":"2024-02-01T18:51:54Z","title":"Towards Efficient Exact Optimization of Language Model Alignment","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.00856","snapshot_observed_at":"2026-08-07T14:41:40.748666Z","title":"Towards efficient exact optimization of language model alignment","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:40.748666Z"},"links":{"cited_paper":"/paper/2402.00856","citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:a6d1aaa5503f8043b21d75147f5089daeed2e77679825a97df62e13c57e028db","observation_id":"d205c502-cb5c-4978-aa94-648277d293ff","resolution":{"observed_at":"2026-08-07T14:41:40.748666Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12599","last_updated":"2025-06-03T02:14:54Z","snapshot_observed_at":"2026-08-15T22:38:53.825110Z","submitted_at":"2025-01-22T02:48:14Z","title":"Kimi k1.5: Scaling Reinforcement Learning with LLMs","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12599","snapshot_observed_at":"2026-08-07T14:41:40.825875Z","title":"Kimi k1.5: Scaling reinforcement learning with llms","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:40.825875Z"},"links":{"cited_paper":"/paper/2501.12599","citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:9695a1ba5c7a045eb7f1632030a1cf80aa9c1bbec78973b42f39c0cebfcd901f","observation_id":"7e5bb8d4-9530-415e-92f3-10efb6609422","resolution":{"observed_at":"2026-08-07T14:41:40.825875Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1412.6980","last_updated":"2017-01-30T01:27:54Z","snapshot_observed_at":"2026-08-17T19:26:44.032537Z","submitted_at":"2014-12-22T13:54:29Z","title":"Adam: A Method for Stochastic Optimization","version":9},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1412.6980","snapshot_observed_at":"2026-08-07T14:41:40.896626Z","title":"Adam: A method for stochastic optimization","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:40.896626Z"},"links":{"cited_paper":"/paper/1412.6980","citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:ee49212a42702682499816f45feb74fb0e10d3bbc0b45dbb82a370e5f4709b01","observation_id":"73bd9f18-90c4-457d-aaab-815e50d0a226","resolution":{"observed_at":"2026-08-07T14:41:40.896626Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:41:41.003001Z","title":"Efficient memory management for large language model serving with pagedattention","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:41.003001Z"},"links":{"citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:23743174e3e7b826ed24336b8445b9708032cc698b99320bffd6a12470ed66ef","observation_id":"dc35b232-f6e9-40ef-8eb2-488e2bb593b7","resolution":{"observed_at":"2026-08-07T14:41:41.003001Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:41:44.062302Z","title":"NuminaMath , 2024","venue":null,"work_id":"c282a1c0-3ba7-46c7-a921-76bf754d11ee","year":2024},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:41.097977Z"},"links":{"citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:307f7b0561cbe37c4f2cbcc9be9e848521598a8590eb630d2f8d9d1fdc6e1d7e","observation_id":"e183c9cf-5934-4b17-8d45-09e7a17989e7","resolution":{"observed_at":"2026-08-07T14:41:44.100819Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.11886","last_updated":"2025-02-17T15:13:29Z","snapshot_observed_at":"2026-08-17T13:27:14.855105Z","submitted_at":"2025-02-17T15:13:29Z","title":"LIMR: Less is More for RL Scaling","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.11886","snapshot_observed_at":"2026-08-07T14:41:41.222191Z","title":"Limr: Less is more for rl scaling","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:41.222191Z"},"links":{"cited_paper":"/paper/2502.11886","citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:64af35359d4654df160d2772f2ea6481374beb6ebac2ca6a9f3801cd06fd6d7a","observation_id":"6f9b73f4-d090-4ab2-93a1-d8eea78a0da2","resolution":{"observed_at":"2026-08-07T14:41:41.222191Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.19393","last_updated":"2025-03-01T06:07:39Z","snapshot_observed_at":"2026-08-17T11:00:39.333660Z","submitted_at":"2025-01-31T18:48:08Z","title":"s1: Simple test-time scaling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.19393","snapshot_observed_at":"2026-08-07T14:41:41.319715Z","title":"s1: Simple test-time scaling","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:41.319715Z"},"links":{"cited_paper":"/paper/2501.19393","citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:0b5913d46f1b83115ced240f6c5436eeb846696ce7b92c92ac3edf8c6934e9cc","observation_id":"307b8061-d6b0-40f8-adf4-7db7c44fbc69","resolution":{"observed_at":"2026-08-07T14:41:41.319715Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.16720","last_updated":"2026-04-30T02:46:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-21T18:04:31Z","title":"OpenAI o1 System Card","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.16720","snapshot_observed_at":"2026-08-07T14:41:41.399866Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:41.399866Z"},"links":{"cited_paper":"/paper/2412.16720","citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:a620441f83699f731d3fe5464ee87f1908be1764151aa5c65b955746df190618","observation_id":"f6015ee9-ea2d-4add-bf2a-9479d800535f","resolution":{"observed_at":"2026-08-07T14:41:41.399866Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06807","last_updated":"2025-02-18T22:21:40Z","snapshot_observed_at":"2026-08-16T21:32:22.938353Z","submitted_at":"2025-02-03T23:00:15Z","title":"Competitive Programming with Large Reasoning Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.06807","snapshot_observed_at":"2026-08-07T14:41:41.482480Z","title":"Competitive Programming with Large Reasoning Models , 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:41.482480Z"},"links":{"cited_paper":"/paper/2502.06807","citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:b1081b4ee5cfa2b0083e3be128a1fd24200f41ef6937b65e5e92b163e7c8064c","observation_id":"0878badd-2b52-442b-baf4-111d63fbec82","resolution":{"observed_at":"2026-08-07T14:41:41.482480Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:41:43.945153Z","title":"QwQ - 32B : Embracing the Power of Reinforcement Learning , March 2025","venue":null,"work_id":"b9607a92-d033-4977-bceb-b77b62003e94","year":2025},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:41.561535Z"},"links":{"citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:f768c18bd13e5676889cd84a0083eddde623c68afb6a9ac184a297b1bc739e00","observation_id":"e4859605-fb3c-40c5-ab4b-5fd7867bf32f","resolution":{"observed_at":"2026-08-07T14:41:43.993712Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:41:41.655714Z","title":"Direct preference optimization: Your language model is secretly a reward model","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:41.655714Z"},"links":{"citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:865b1a9b2d785808486b72b7445ed31735d9e5951a7ed733150dba44547ed0ee","observation_id":"158a710d-d6a8-4309-8aa9-5b8ac4361a2e","resolution":{"observed_at":"2026-08-07T14:41:41.655714Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10490","last_updated":"2025-06-29T05:43:22Z","snapshot_observed_at":"2026-08-16T13:34:16.091068Z","submitted_at":"2024-07-15T07:30:28Z","title":"Learning Dynamics of LLM Finetuning","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.10490","snapshot_observed_at":"2026-08-07T14:41:41.745767Z","title":"Learning dynamics of llm finetuning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:41.745767Z"},"links":{"cited_paper":"/paper/2407.10490","citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:7957434c2b21362cf432c8b55bf757dca19d164f0d214a69a51f2e70d07f3426","observation_id":"cd2a7a68-530a-49df-9b75-a23b2a061db7","resolution":{"observed_at":"2026-08-07T14:41:41.745767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.19256","last_updated":"2024-10-02T04:01:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-28T06:20:03Z","title":"HybridFlow: A Flexible and Efficient RLHF Framework","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.19256","snapshot_observed_at":"2026-08-07T14:41:41.846490Z","title":"Hybridflow: A flexible and efficient rlhf framework","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:41.846490Z"},"links":{"cited_paper":"/paper/2409.19256","citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:0e721dde3ec5289dc0402d4dc28b6a755dfe22accfb366f5897da3cfc1a3cc2b","observation_id":"4aeecf9d-8acd-4dc7-afc4-42c11985a68d","resolution":{"observed_at":"2026-08-07T14:41:41.846490Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:41:43.756396Z","title":"Policy Gradient Methods for Reinforcement Learning with Function Approximation","venue":null,"work_id":"41a63cb7-d41b-4a70-b971-3e3af8d221a0","year":1999},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:41.946416Z"},"links":{"citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:51fa928128602cb745223df1a4d86197b64f2890350372c322ef0d4664c6243e","observation_id":"42448170-c7f8-432e-835a-fada56558a22","resolution":{"observed_at":"2026-08-07T14:41:43.846736Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:41:43.607599Z","title":"Chain-of-thought prompting elicits reasoning in large language models","venue":null,"work_id":"94999042-2b19-4031-b393-7a1fda93314e","year":2022},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:42.046397Z"},"links":{"citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:f0cee4c9fd1b085c635d188011c8250ec01c2215d7330678e77bd30524553735","observation_id":"3e902f7e-4ce1-41cb-b93e-cbbfac471a4d","resolution":{"observed_at":"2026-08-07T14:41:43.669174Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.23123","last_updated":"2025-03-04T06:22:40Z","snapshot_observed_at":"2026-08-16T13:04:30.008201Z","submitted_at":"2024-10-30T15:31:54Z","title":"On Memorization of Large Language Models in Logical Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.23123","snapshot_observed_at":"2026-08-07T14:41:42.174397Z","title":"On memorization of large language models in logical reasoning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:42.174397Z"},"links":{"cited_paper":"/paper/2410.23123","citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:fc9272e1d2382a1788eb0310583ce0f74e20608679d450a3a4b27899000fa097","observation_id":"18575e88-56ee-489f-9617-fde83ef0a83a","resolution":{"observed_at":"2026-08-07T14:41:42.174397Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14768","last_updated":"2025-02-20T17:49:26Z","snapshot_observed_at":"2026-08-15T06:02:45.207770Z","submitted_at":"2025-02-20T17:49:26Z","title":"Logic-RL: Unleashing LLM Reasoning with Rule-Based Reinforcement Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14768","snapshot_observed_at":"2026-08-07T14:41:42.276120Z","title":"Logic-rl: Unleashing llm reasoning with rule-based reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:42.276120Z"},"links":{"cited_paper":"/paper/2502.14768","citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:86263a01a68197d50881b4df9543ab11893bca32b44c84cf829749ad750ed6b2","observation_id":"ed7b0955-9280-4627-9ee8-773c114f2c72","resolution":{"observed_at":"2026-08-07T14:41:42.276120Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-08-17T18:50:07.059564Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-07T14:41:42.361609Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:42.361609Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:2bccb0372a4b0072052e91e83f2389146bdf81884c13a1531401bb49fa7e4fb0","observation_id":"4d295d10-5b4a-4dd3-bec7-ac4081772af4","resolution":{"observed_at":"2026-08-07T14:41:42.361609Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12122","last_updated":"2024-09-18T16:45:37Z","snapshot_observed_at":"2026-08-17T18:51:13.219936Z","submitted_at":"2024-09-18T16:45:37Z","title":"Qwen2.5-Math Technical Report: Toward Mathematical Expert Model via Self-Improvement","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12122","snapshot_observed_at":"2026-08-07T14:41:42.429652Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:42.429652Z"},"links":{"cited_paper":"/paper/2409.12122","citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:785b6c3c4b581488226eb6a70216ac1ee186ccaefd3f4c8cc48443c6c8cc03ed","observation_id":"e130b08a-299f-443b-a932-1891b1f24973","resolution":{"observed_at":"2026-08-07T14:41:42.429652Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.03387","last_updated":"2025-07-29T16:23:02Z","snapshot_observed_at":"2026-08-08T04:13:22.884923Z","submitted_at":"2025-02-05T17:23:45Z","title":"LIMO: Less is More for Reasoning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.03387","snapshot_observed_at":"2026-08-07T14:41:42.502205Z","title":"LIMO : Less is More for Reasoning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:42.502205Z"},"links":{"cited_paper":"/paper/2502.03387","citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:fc1878aeb09f91c255633951b94f612eb12f1e8af5d88c9eaafafe778dc56557","observation_id":"0e849a46-9f46-4624-a898-8e79d93d41b8","resolution":{"observed_at":"2026-08-07T14:41:42.502205Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.03373","last_updated":"2025-02-05T17:13:32Z","snapshot_observed_at":"2026-08-17T16:40:23.847561Z","submitted_at":"2025-02-05T17:13:32Z","title":"Demystifying Long Chain-of-Thought Reasoning in LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.03373","snapshot_observed_at":"2026-08-07T14:41:42.610566Z","title":"Demystifying long chain-of-thought reasoning in llms, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:42.610566Z"},"links":{"cited_paper":"/paper/2502.03373","citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:869e1af63726715cd5904ed253bbf6466f74ca22352530d5f108e15b6ddb34e7","observation_id":"876f7086-f4b9-43fe-933f-97658e4fe52c","resolution":{"observed_at":"2026-08-07T14:41:42.610566Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.14476","last_updated":"2025-05-20T01:37:34Z","snapshot_observed_at":"2026-08-02T01:40:54.187278Z","submitted_at":"2025-03-18T17:49:06Z","title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.14476","snapshot_observed_at":"2026-08-07T14:41:42.685105Z","title":"Dapo: An open-source llm reinforcement learning system at scale","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:42.685105Z"},"links":{"cited_paper":"/paper/2503.14476","citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:3f10f3c256c1efd44922659232a2741355150f511abe302a5ae322369d70ba19","observation_id":"63cbc4b7-953c-499d-986d-d3fc73d92757","resolution":{"observed_at":"2026-08-07T14:41:42.685105Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.18892","last_updated":"2025-08-06T08:42:32Z","snapshot_observed_at":"2026-07-06T20:57:57.039376Z","submitted_at":"2025-03-24T17:06:10Z","title":"SimpleRL-Zoo: Investigating and Taming Zero Reinforcement Learning for Open Base Models in the Wild","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.18892","snapshot_observed_at":"2026-08-07T14:41:42.792713Z","title":"Simplerl-zoo: Investigating and taming zero reinforcement learning for open base models in the wild","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:42.792713Z"},"links":{"cited_paper":"/paper/2503.18892","citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:05c68382fa17f3ce05d776bbd7d66df872539f34779d53fc94394617e549704b","observation_id":"8016a355-3219-458e-862c-f60374c936ef","resolution":{"observed_at":"2026-08-07T14:41:42.792713Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.07912","last_updated":"2025-08-07T23:50:47Z","snapshot_observed_at":"2026-08-17T19:03:31.231829Z","submitted_at":"2025-04-10T17:15:53Z","title":"Echo Chamber: RL Post-training Amplifies Behaviors Learned in Pretraining","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.07912","snapshot_observed_at":"2026-08-07T14:41:42.861182Z","title":"Echo chamber: Rl post-training amplifies behaviors learned in pretraining","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","version":3},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-07T14:41:42.861182Z"},"links":{"cited_paper":"/paper/2504.07912","citing_paper":"/paper/2505.17988"},"observation_digest":"sha256:bedfc1a4a117a943f0b668d8132103a6aafaaa0e6fc40d0a7402a3dc11578032","observation_id":"b1265abc-8c31-453d-8e78-17fdf1016739","resolution":{"observed_at":"2026-08-07T14:41:42.861182Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2505.17988","last_updated":"2025-08-05T11:46:13Z","latest_version":3,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-09T20:50:22.177201Z","submitted_at":"2025-05-23T14:55:22Z","title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning"},"reference_resolution":{"displayed":39,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":30,"verified_exact":0,"verified_fuzzy":9},"total_outbound_references":39},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"thesis":"As of 17 August 2026, this Paper Citation Record lists 39 of 39 outbound references and 1 inbound Pith citation observation for arXiv:2505.17988."}