{"as_of":"2026-08-21T13:31:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:1ecfffa81b4674cff867fc7173dbcf4f2672311e165e75eb8f83b2253d557425","coverage":[{"denominator":43,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":43,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:32:31.776781Z","state":"measured"},{"denominator":46,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":46,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-21T06:32:19.484+00:00","state":"measured"},{"denominator":3,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":3,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T16:07:30.092931Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-12T08:40:41.514103Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"cited_work":{"arxiv_id":"2506.02553","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.02553","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Response-level rewards are all you need for online reinforcement learning in llms: A mathematical perspective","venue":null,"work_id":"bd00da2d-e4a8-4dab-9a51-19d8603c4907","year":2025},"citing_paper":{"arxiv_id":"2503.09567","last_updated":"2025-07-18T15:57:54Z","snapshot_observed_at":"2026-08-08T22:33:20.124926Z","submitted_at":"2025-03-12T17:35:03Z","title":"Towards Reasoning Era: A Survey of Long Chain-of-Thought for Reasoning Large Language Models","version":5},"reference_index":252,"source":"pdf_text","source_observed_at":"2026-05-12T08:40:40.910461Z"},"links":{"cited_paper":"/paper/2506.02553","citing_paper":"/paper/2503.09567"},"observation_digest":"sha256:68beb9aa2ab53ae2340aeba31fa3d0a6b95dc15994026eb4c597362fff6f94fb","observation_id":"458b6a9e-4f12-408c-853b-786e6f2ba5ed","resolution":{"observed_at":"2026-05-12T08:40:41.517098Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.02553","snapshot_observed_at":"2026-08-04T16:07:30.092931Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.16679","last_updated":"2025-09-20T13:11:28Z","snapshot_observed_at":"2026-08-15T19:34:29.390362Z","submitted_at":"2025-09-20T13:11:28Z","title":"Reinforcement Learning Meets Large Language Models: A Survey of Advancements and Applications Across the LLM Lifecycle","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-04T16:07:30.092931Z"},"links":{"cited_paper":"/paper/2506.02553","citing_paper":"/paper/2509.16679"},"observation_digest":"sha256:f3de731555e6f6c808b32208557f78ea4aafc5faf8dfc17e8dc936a86d6393d6","observation_id":"cff38d40-0384-4dfd-b50f-4b7a1be48da5","resolution":{"observed_at":"2026-08-04T16:07:30.092931Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"cited_work":{"arxiv_id":"2506.02553","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.02553","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Response-level rewards are all you need for online reinforcement learning in llms: A mathematical perspective","venue":null,"work_id":"bd00da2d-e4a8-4dab-9a51-19d8603c4907","year":2025},"citing_paper":{"arxiv_id":"2605.10218","last_updated":"2026-05-11T08:58:40Z","snapshot_observed_at":"2026-08-15T13:10:53.017112Z","submitted_at":"2026-05-11T08:58:40Z","title":"Relative Score Policy Optimization for Diffusion Language Models","version":1},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-05-12T03:47:42.196931Z"},"links":{"cited_paper":"/paper/2506.02553","citing_paper":"/paper/2605.10218"},"observation_digest":"sha256:01a8d7a40821f696c0d510d8c4dd4acfe1567d84de8b4e10225e2452edb9a70c","observation_id":"389afa80-1000-430a-9628-73a046081c5c","resolution":{"observed_at":"2026-05-12T06:56:29.690474Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.02553/citation-record","integrity":"/paper/2506.02553/integrity","json":"/paper/2506.02553/citation-record.json","paper":"/paper/2506.02553"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-08-15T14:02:47.366139Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-07T11:32:28.119855Z","title":"Gpt-4o system card.arXiv preprint arXiv:2410.21276, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:28.119855Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:378232681f266d68c8d1c363e23ad15874b65a56e84027f0c5462a14e6ae8697","observation_id":"a1610d62-8496-4528-b307-6534336b9c74","resolution":{"observed_at":"2026-08-07T11:32:28.119855Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-08-20T18:27:04.837880Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-07T11:32:28.178328Z","title":"Gemini: a family of highly capable multimodal models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:28.178328Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:bd96628573a35824d3a2bab64abb80b4c920b3dc070dfcb805fca5e05b63ad46","observation_id":"073987cd-41cf-4941-a263-ca7b0bcbdc3c","resolution":{"observed_at":"2026-08-07T11:32:28.178328Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-08-17T18:50:07.059564Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-07T11:32:28.272952Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:28.272952Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:0a76fe503ebb75034a66b746b83c8b8e621ff98ca50b7cb053e7c756f3ffcb69","observation_id":"d847781b-9233-4555-8232-c1e126abbad6","resolution":{"observed_at":"2026-08-07T11:32:28.272952Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:32:33.247829Z","title":"The amazon nova family of models: Technical report and model card","venue":null,"work_id":"a2f3d7c8-274d-4c98-a64e-1c94d9a6fc72","year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:28.391741Z"},"links":{"citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:7f17ced61b7306f0cff0379bbc9d1a4294a4e768bb8407d7757f417afbbf199e","observation_id":"fac76498-5a66-438f-8048-394162622066","resolution":{"observed_at":"2026-08-07T11:32:33.438977Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.16720","last_updated":"2026-04-30T02:46:40Z","snapshot_observed_at":"2026-08-19T11:46:55.171293Z","submitted_at":"2024-12-21T18:04:31Z","title":"OpenAI o1 System Card","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.16720","snapshot_observed_at":"2026-08-07T11:32:28.487274Z","title":"Openai o1 system card.arXiv preprint arXiv:2412.16720, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:28.487274Z"},"links":{"cited_paper":"/paper/2412.16720","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:24a805b1af4caf8a83164bb9f358b0265f7c195be4d3c489348c198f7ad428a1","observation_id":"8cfe9e74-aa5e-4339-a953-161932231c2a","resolution":{"observed_at":"2026-08-07T11:32:28.487274Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-08-15T12:33:55.451951Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-07T11:32:28.616012Z","title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning.arXiv preprint arXiv:2501.12948, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:28.616012Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:219a2fc444994e224c347cb7dfa995fe92c3d78a39e39573cf57c33071ec6262","observation_id":"838fd653-fbef-4b82-b5c1-7cc2b920e801","resolution":{"observed_at":"2026-08-07T11:32:28.616012Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-08-20T07:04:06.309989Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-08-07T11:32:28.675759Z","title":"Proximal policy optimization algorithms.arXiv preprint arXiv:1707.06347, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:28.675759Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:3d09bfa035c55a6608a1abae6f2ba5677c3eae04cb1930a770535d42e6c0f8ef","observation_id":"206e65fa-4c05-4e23-83ff-e6b70f85acf8","resolution":{"observed_at":"2026-08-07T11:32:28.675759Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:32:28.759565Z","title":"Learning to summarize with human feedback.Advances in neural information processing systems, 33:3008–3021, 2020","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:28.759565Z"},"links":{"citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:f3c6268ec6be43aed24d1c3144238fd411fe6d1360ce57e8ab62ecb2310da085","observation_id":"7a22d8ee-ed50-44dc-9392-7f7f9866d0bb","resolution":{"observed_at":"2026-08-07T11:32:28.759565Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:32:28.849369Z","title":"Training language models to follow instructions with human feedback.Advances in neural information processing systems, 35:27730–27744, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:28.849369Z"},"links":{"citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:096fc6862ac07a8e7b41728ff9f9bfcf394f5d2091880724f7d7638e926f9728","observation_id":"db41d348-05c2-4c17-b586-f1c328306197","resolution":{"observed_at":"2026-08-07T11:32:28.849369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2204.05862","last_updated":"2022-04-12T15:02:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-04-12T15:02:38Z","title":"Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.05862","snapshot_observed_at":"2026-08-07T11:32:28.936948Z","title":"Training a helpful and harmless assistant with reinforcement learning from human feedback.arXiv preprint arXiv:2204.05862, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:28.936948Z"},"links":{"cited_paper":"/paper/2204.05862","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:08007804679ee26448900d6f1e6b76717562f7678d1b0ce3a870a690dbcecde1","observation_id":"fa568fdd-33ba-47a7-8f3a-c6a07f2bdd31","resolution":{"observed_at":"2026-08-07T11:32:28.936948Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10505","last_updated":"2024-05-16T02:22:23Z","snapshot_observed_at":"2026-08-19T17:31:23.637137Z","submitted_at":"2023-10-16T15:25:14Z","title":"ReMax: A Simple, Effective, and Efficient Reinforcement Learning Method for Aligning Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10505","snapshot_observed_at":"2026-08-07T11:32:29.011230Z","title":"Remax: A sim- ple, effective, and efficient reinforcement learning method for aligning large language models.arXiv preprint arXiv:2310.10505, 2023","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:29.011230Z"},"links":{"cited_paper":"/paper/2310.10505","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:894aea899f54496e83cf1bc7574c0d0137a8e80bf5a81ca7b5ba701c3a56da1b","observation_id":"b3b842df-d41e-4957-991c-f0ea6963c5f2","resolution":{"observed_at":"2026-08-07T11:32:29.011230Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-07T11:32:29.157791Z","title":"Deepseekmath: Pushing the limits of mathematical reasoning in open language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:29.157791Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:f6dc8dc5ed8a0eab70a2358f45933a4a7f7593f21ad0b975506496b817f59e2c","observation_id":"92d7bb58-110a-4dc1-b1a8-d6fdf26249d2","resolution":{"observed_at":"2026-08-07T11:32:29.157791Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14740","last_updated":"2024-02-26T18:26:25Z","snapshot_observed_at":"2026-08-09T14:30:33.899591Z","submitted_at":"2024-02-22T17:52:34Z","title":"Back to Basics: Revisiting REINFORCE Style Optimization for Learning from Human Feedback in LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.14740","snapshot_observed_at":"2026-08-07T11:32:29.285863Z","title":"Back to basics: Revisiting reinforce style optimization for learning from human feedback in llms.arXiv preprint arXiv:2402.14740, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:29.285863Z"},"links":{"cited_paper":"/paper/2402.14740","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:c8be7fee467e26352be1c2dd7b1af375c9beb76c4e0116bf37bbfdd42acb56ab","observation_id":"ac4235ba-1686-4383-89d7-670ef940fb6a","resolution":{"observed_at":"2026-08-07T11:32:29.285863Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.18922","last_updated":"2025-05-21T15:34:02Z","snapshot_observed_at":"2026-08-16T17:38:22.018669Z","submitted_at":"2024-04-29T17:58:30Z","title":"DPO Meets PPO: Reinforced Token Optimization for RLHF","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.18922","snapshot_observed_at":"2026-08-07T11:32:29.379167Z","title":"Dpo meets ppo: Reinforced token optimization for rlhf.arXiv preprint arXiv:2404.18922, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:29.379167Z"},"links":{"cited_paper":"/paper/2404.18922","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:fb500401b39019db783c7c47c96c84ad4b547a9dd403d795e452f5156017f04d","observation_id":"c31fd329-18ae-4c4a-af35-5ef3063ac485","resolution":{"observed_at":"2026-08-07T11:32:29.379167Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.00782","last_updated":"2024-02-01T17:10:35Z","snapshot_observed_at":"2026-08-20T14:41:03.082523Z","submitted_at":"2024-02-01T17:10:35Z","title":"Dense Reward for Free in Reinforcement Learning from Human Feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.00782","snapshot_observed_at":"2026-08-07T11:32:29.474618Z","title":"Dense reward for free in reinforcement learning from human feedback.arXiv preprint arXiv:2402.00782, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:29.474618Z"},"links":{"cited_paper":"/paper/2402.00782","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:1df4785cc82084c127c58a14e6379aa9d36362896baea35993d7298d55d17e4b","observation_id":"3d4739b4-8fbb-4f16-9d85-ca8afd503afd","resolution":{"observed_at":"2026-08-07T11:32:29.474618Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01456","last_updated":"2025-09-26T09:25:31Z","snapshot_observed_at":"2026-08-20T04:08:34.710482Z","submitted_at":"2025-02-03T15:43:48Z","title":"Process Reinforcement through Implicit Rewards","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.01456","snapshot_observed_at":"2026-08-07T11:32:29.517440Z","title":"Process reinforcement through implicit rewards.arXiv preprint arXiv:2502.01456, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:29.517440Z"},"links":{"cited_paper":"/paper/2502.01456","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:d2d1fd4f8910b6b14326b5f8bcb1780235ce6a2cca82006ab27a22fc8276681b","observation_id":"9dc7d6f1-aa41-45d1-a570-87e7f284de1d","resolution":{"observed_at":"2026-08-07T11:32:29.517440Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.08302","last_updated":"2025-09-11T10:17:06Z","snapshot_observed_at":"2026-08-19T14:41:11.334374Z","submitted_at":"2024-11-13T02:45:21Z","title":"RED: Unleashing Token-Level Rewards from Holistic Feedback via Reward Redistribution","version":2},"cited_work":{"arxiv_id":"2411.08302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2411.08302","snapshot_observed_at":"2026-08-07T11:32:32.239050Z","title":"RED: Unleashing Token-Level Rewards from Holistic Feedback via Reward Redistribution","venue":"cs.CL","work_id":"b790476c-97f6-4051-8e20-05fa884acb61","year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:29.571593Z"},"links":{"cited_paper":"/paper/2411.08302","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:15fe3e00c3ab1964daf552f089c7466a4cc87c67bea41593d61ab3f89ff4d1ba","observation_id":"d1e21099-45e5-456f-9a22-a89f33de763d","resolution":{"observed_at":"2026-08-07T11:32:32.289654Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.10719","last_updated":"2024-10-10T08:30:17Z","snapshot_observed_at":"2026-08-20T11:04:33.167279Z","submitted_at":"2024-04-16T16:51:53Z","title":"Is DPO Superior to PPO for LLM Alignment? A Comprehensive Study","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.10719","snapshot_observed_at":"2026-08-07T11:32:29.652497Z","title":"Is dpo superior to ppo for llm alignment? a comprehensive study.arXiv preprint arXiv:2404.10719, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:29.652497Z"},"links":{"cited_paper":"/paper/2404.10719","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:402a3792a67bd16a3fc364824cb09ab07c59a90659f9ea77ea0734d1af316dc7","observation_id":"226e8b7d-71ea-4786-aa9b-ea8cb8881666","resolution":{"observed_at":"2026-08-07T11:32:29.652497Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:32:33.051822Z","title":"Reinforcement learning: An introduction","venue":null,"work_id":"3e8eb64a-b069-4cc4-a72e-787b7e3bf1a0","year":2018},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:29.736929Z"},"links":{"citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:77622e544127facb3ca84f2b077577a4364b12a1a7ea481bdd2b8dba2d0decde","observation_id":"0f06674a-aab8-4c2f-ace5-3d413506452c","resolution":{"observed_at":"2026-08-07T11:32:33.106505Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2503.22244","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:32:32.108585Z","title":"Analysis of on-policy policy gradient methods under the distribution mismatch.arXiv preprint arXiv:2503.22244, 2025","venue":null,"work_id":"4e6b86a9-119a-483d-bbfc-8eade381a577","year":2025},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:29.838106Z"},"links":{"citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:6bd2eac6407c520add7ef34b92d03dea59ab18d053a38f21e4ceb91d546c64d6","observation_id":"e3eae744-e475-416a-b86c-3b11a18d485e","resolution":{"observed_at":"2026-08-07T11:32:32.175493Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:32:32.957520Z","title":"High-dimensional continu- ous control using generalized advantage estimation, 2018","venue":null,"work_id":"e523ba1c-ae48-4339-9d84-97ce562cfb00","year":2018},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:29.935825Z"},"links":{"citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:1fd43fcc6c2ae41b305ec85f8adf06468889e3a0fda24b2e278adb89f930e3c3","observation_id":"ebc8704f-7ad7-42cb-b105-420a41cba4df","resolution":{"observed_at":"2026-08-07T11:32:32.986611Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.01679","last_updated":"2025-06-03T20:51:06Z","snapshot_observed_at":"2026-08-21T11:41:50.405180Z","submitted_at":"2024-10-02T15:49:30Z","title":"VinePPO: Refining Credit Assignment in RL Training of LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.01679","snapshot_observed_at":"2026-08-07T11:32:30.026404Z","title":"Vineppo: Unlocking rl potential for llm reasoning through refined credit assignment.arXiv preprint arXiv:2410.01679, 2024","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:30.026404Z"},"links":{"cited_paper":"/paper/2410.01679","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:2541f4ee49652bf6cd413d74a7f82ba705d143f16c248e56c75d9584859d00d4","observation_id":"229e0672-ac94-41c0-aad3-7d737231e735","resolution":{"observed_at":"2026-08-07T11:32:30.026404Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:32:30.149593Z","title":"Gonzalez, Hao Zhang, and Ion Stoica","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:30.149593Z"},"links":{"citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:8e6bb27b0db208fe3b4dc634f0d3219fedea9ed6b1071ebe012e08fc49aec39c","observation_id":"faea92d1-0529-408c-8cf3-62950293087a","resolution":{"observed_at":"2026-08-07T11:32:30.149593Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:32:32.819233Z","title":"Direct preference optimization: Your language model is secretly a reward model.Advances in Neural Information Processing Systems, 36:53728–53741, 2023","venue":null,"work_id":"e4c6c81e-3f88-4bdd-b339-7a6e8bb1a6ca","year":2023},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:30.236850Z"},"links":{"citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:e5d4ff43a378c834c542dbb228b570edc1762827f52aef29d4c9e75a8a3e2546","observation_id":"8d855c2e-f274-4eab-97d1-d23f6041ab13","resolution":{"observed_at":"2026-08-07T11:32:32.869318Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.15478","last_updated":"2025-03-19T17:55:08Z","snapshot_observed_at":"2026-08-19T17:31:41.105938Z","submitted_at":"2025-03-19T17:55:08Z","title":"SWEET-RL: Training Multi-Turn LLM Agents on Collaborative Reasoning Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.15478","snapshot_observed_at":"2026-08-07T11:32:30.314881Z","title":"Sweet-rl: Training multi-turn llm agents on collaborative reasoning tasks.arXiv preprint arXiv:2503.15478, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:30.314881Z"},"links":{"cited_paper":"/paper/2503.15478","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:8c4b9efeed93a5ca2a36b42aa60acc82f80331f53b5cf32e6b465db701996329","observation_id":"3d37a5d8-7487-4031-94a2-c63c0b08b245","resolution":{"observed_at":"2026-08-07T11:32:30.314881Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.14655","last_updated":"2024-12-02T12:37:46Z","snapshot_observed_at":"2026-08-17T05:06:22.790424Z","submitted_at":"2024-05-23T14:53:54Z","title":"Multi-turn Reinforcement Learning from Preference Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.14655","snapshot_observed_at":"2026-08-07T11:32:30.406644Z","title":"Multi-turn reinforcement learning from preference human feedback","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:30.406644Z"},"links":{"cited_paper":"/paper/2405.14655","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:4ad14da66cd39df95539217a3118e785edc92073719e048e978260dc97f36d95","observation_id":"06c03cc1-5efd-4109-ad8b-30f0011fe3dc","resolution":{"observed_at":"2026-08-07T11:32:30.406644Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.18232","last_updated":"2023-11-30T03:59:31Z","snapshot_observed_at":"2026-08-16T14:38:59.314544Z","submitted_at":"2023-11-30T03:59:31Z","title":"LMRL Gym: Benchmarks for Multi-Turn Reinforcement Learning with Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.18232","snapshot_observed_at":"2026-08-07T11:32:30.490860Z","title":"Lmrl gym: Benchmarks for multi-turn reinforcement learning with language models.arXiv preprint arXiv:2311.18232, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:30.490860Z"},"links":{"cited_paper":"/paper/2311.18232","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:fdd45de215633186806ea22b2638a641289af6153ca365c67760302582009868","observation_id":"4ef9d1d0-0259-4636-8775-80a7ff65f737","resolution":{"observed_at":"2026-08-07T11:32:30.490860Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.19446","last_updated":"2024-02-29T18:45:56Z","snapshot_observed_at":"2026-08-19T03:42:50.568302Z","submitted_at":"2024-02-29T18:45:56Z","title":"ArCHer: Training Language Model Agents via Hierarchical Multi-Turn RL","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.19446","snapshot_observed_at":"2026-08-07T11:32:30.549942Z","title":"Archer: Training language model agents via hierarchical multi-turn rl.arXiv preprint arXiv:2402.19446, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:30.549942Z"},"links":{"cited_paper":"/paper/2402.19446","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:77269dcaf308a6f315ce85d773d3161180019b2191112570cba0a948944609e4","observation_id":"f04a9702-4b69-44c0-ac3b-86915a75c762","resolution":{"observed_at":"2026-08-07T11:32:30.549942Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.16145","last_updated":"2024-12-25T18:54:02Z","snapshot_observed_at":"2026-08-18T16:03:34.886701Z","submitted_at":"2024-12-20T18:49:45Z","title":"Offline Reinforcement Learning for LLM Multi-Step Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.16145","snapshot_observed_at":"2026-08-07T11:32:30.615635Z","title":"Offline reinforcement learning for llm multi-step reasoning.arXiv preprint arXiv:2412.16145, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:30.615635Z"},"links":{"cited_paper":"/paper/2412.16145","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:34cbf7ab3e7132885b56b5a399194616c5787ad61b62dbee8894e4230472fce7","observation_id":"40d8a2c3-061f-4d2d-ab59-133b144a810e","resolution":{"observed_at":"2026-08-07T11:32:30.615635Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.11221","last_updated":"2025-06-23T05:32:12Z","snapshot_observed_at":"2026-08-20T13:30:02.735882Z","submitted_at":"2025-02-16T17:54:57Z","title":"PlanGenLLMs: A Modern Survey of LLM Planning Capabilities","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.11221","snapshot_observed_at":"2026-08-07T11:32:30.682383Z","title":"Plangenllms: A modern survey of llm planning capabilities.arXiv preprint arXiv:2502.11221, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:30.682383Z"},"links":{"cited_paper":"/paper/2502.11221","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:bc0e263b655a69b40da23521576e332dd3b7d6522fea2feefac18efe63ad5c45","observation_id":"46c7a99b-9411-4c48-9a88-2d440dd9bf5d","resolution":{"observed_at":"2026-08-07T11:32:30.682383Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14762","last_updated":"2024-11-05T16:40:21Z","snapshot_observed_at":"2026-08-16T14:16:09.123077Z","submitted_at":"2024-02-22T18:21:59Z","title":"MT-Bench-101: A Fine-Grained Benchmark for Evaluating Large Language Models in Multi-Turn Dialogues","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.14762","snapshot_observed_at":"2026-08-07T11:32:30.761569Z","title":"Mt-bench-101: A fine-grained benchmark for evaluating large language models in multi-turn dialogues.arXiv preprint arXiv:2402.14762, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:30.761569Z"},"links":{"cited_paper":"/paper/2402.14762","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:944997a737d4a2a33afb4e404018b1321a27ec4c8451326f044587c4affc7883","observation_id":"d316c10c-5751-4cb2-a82f-c5351ec05453","resolution":{"observed_at":"2026-08-07T11:32:30.761569Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:32:32.658344Z","title":"Interactive evaluation for medical LLMs via task- oriented dialogue system","venue":null,"work_id":"9ee545f8-f127-4b66-b3a6-79977eaa464c","year":2025},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:30.841930Z"},"links":{"citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:52b1920e0720a04bf8676b2b4f4b2938f811b67023b99fbd43c4928d25952061","observation_id":"4c96c6b6-cb23-4153-9448-1ad05c0b0ae9","resolution":{"observed_at":"2026-08-07T11:32:32.721910Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:32:30.927433Z","title":"Let’s verify step by step, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:30.927433Z"},"links":{"citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:18b7e163ee9ac41492b6899213ac9f53c23bad02da374168a74f170d027bfb45","observation_id":"7ed50a14-fea9-4537-9fe4-15c531531cc2","resolution":{"observed_at":"2026-08-07T11:32:30.927433Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:32:32.480645Z","title":"Unraveling rlhf and its variants: Progress and practical engineering insights","venue":null,"work_id":"80c0526d-a3f2-4b8b-a77e-b439ed2f7bb0","year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:31.056825Z"},"links":{"citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:c260f291f88e8313b2c9ed40c83721ee5fa800751cb3ee64de24291f561e24f9","observation_id":"7a0a2145-8133-4678-ac20-be71fc5c50ae","resolution":{"observed_at":"2026-08-07T11:32:32.554729Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2211.14275","last_updated":"2022-11-25T18:19:44Z","snapshot_observed_at":"2026-08-01T02:16:43.109337Z","submitted_at":"2022-11-25T18:19:44Z","title":"Solving math word problems with process- and outcome-based feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2211.14275","snapshot_observed_at":"2026-08-07T11:32:31.151918Z","title":"Solving math word problems with process-and outcome-based feedback","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:31.151918Z"},"links":{"cited_paper":"/paper/2211.14275","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:123e0996ee64354477b3242f1683f64e33d68336f93d06fc9c3134ef810a59b0","observation_id":"c4d243be-85b3-4deb-8dd7-850e0b14cfb7","resolution":{"observed_at":"2026-08-07T11:32:31.151918Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:32:31.223415Z","title":"Fine-grained human feedback gives better rewards for language model training.Advances in Neural Information Processing Systems, 36:59008–59033, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:31.223415Z"},"links":{"citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:cdaffc6fbce2852d76278db81bebfea113e1cdc281b3b1bc96dc4967e150966e","observation_id":"ce13e47a-3bf7-42d3-bbf8-efa2c0227fb7","resolution":{"observed_at":"2026-08-07T11:32:31.223415Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07382","last_updated":"2024-02-19T18:19:20Z","snapshot_observed_at":"2026-08-16T14:27:36.266197Z","submitted_at":"2024-01-14T22:05:11Z","title":"Beyond Sparse Rewards: Enhancing Reinforcement Learning with Language Model Critique in Text Generation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.07382","snapshot_observed_at":"2026-08-07T11:32:31.310922Z","title":"Beyond sparse re- wards: Enhancing reinforcement learning with language model critique in text generation.arXiv preprint arXiv:2401.07382, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:31.310922Z"},"links":{"cited_paper":"/paper/2401.07382","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:fa3cc49d6d67dd65a07d881ef64cc5c773a5fb14d5e50e9f9e5f6d13c4a5a15b","observation_id":"86e6e972-8ff4-4c20-a920-2dfcddb4ecf9","resolution":{"observed_at":"2026-08-07T11:32:31.310922Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.16574","last_updated":"2024-12-08T14:22:13Z","snapshot_observed_at":"2026-08-19T09:52:23.844650Z","submitted_at":"2024-07-23T15:27:37Z","title":"TLCR: Token-Level Continuous Reward for Fine-grained Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.16574","snapshot_observed_at":"2026-08-07T11:32:31.365581Z","title":"Tlcr: Token-level continuous reward for fine-grained reinforcement learning from human feedback.arXiv preprint arXiv:2407.16574, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:31.365581Z"},"links":{"cited_paper":"/paper/2407.16574","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:7ae8a7c19eed279c3f3bce7d1a0c9368e90fe1a9a7e9db4f619bdef1b57ed87d","observation_id":"70da07b1-0cf1-44f1-ba5c-0c3613c1b6ab","resolution":{"observed_at":"2026-08-07T11:32:31.365581Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.08935","last_updated":"2024-02-19T14:07:53Z","snapshot_observed_at":"2026-08-19T00:49:04.283297Z","submitted_at":"2023-12-14T13:41:54Z","title":"Math-Shepherd: Verify and Reinforce LLMs Step-by-step without Human Annotations","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.08935","snapshot_observed_at":"2026-08-07T11:32:31.431529Z","title":"Math- shepherd: Verify and reinforce llms step-by-step without human annotations.arXiv preprint arXiv:2312.08935, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:31.431529Z"},"links":{"cited_paper":"/paper/2312.08935","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:f6fabfa7e3771982eb2bd8c79c247338ac3ad797c1e276fa09d4081a47de3337","observation_id":"005c73bf-7ead-4a34-a89f-163671031cb4","resolution":{"observed_at":"2026-08-07T11:32:31.431529Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06781","last_updated":"2025-02-10T18:57:29Z","snapshot_observed_at":"2026-08-16T03:56:33.383172Z","submitted_at":"2025-02-10T18:57:29Z","title":"Exploring the Limit of Outcome Reward for Learning Mathematical Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.06781","snapshot_observed_at":"2026-08-07T11:32:31.527587Z","title":"Exploring the limit of outcome reward for learning mathematical reasoning.arXiv preprint arXiv:2502.06781, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:31.527587Z"},"links":{"cited_paper":"/paper/2502.06781","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:a62f5b6354aa0792e5dbb9202b6f90f0a06ef86950c98f2048a779a6397baf2f","observation_id":"867e3e6f-9b30-47c1-b77b-cd076a11bcef","resolution":{"observed_at":"2026-08-07T11:32:31.527587Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:32:31.601560Z","title":"Chatbot arena: An open platform for evaluating llms by human preference","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:31.601560Z"},"links":{"citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:11446b928649b1dfc815a8e1026fc59cd4752c888609986a6113cd757916928f","observation_id":"eb286a80-c534-4ce2-98f5-35563562af88","resolution":{"observed_at":"2026-08-07T11:32:31.601560Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.13006","last_updated":"2025-03-30T17:59:47Z","snapshot_observed_at":"2026-08-18T13:17:17.701561Z","submitted_at":"2024-08-23T11:49:01Z","title":"Systematic Evaluation of LLM-as-a-Judge in LLM Alignment Tasks: Explainable Metrics and Diverse Prompt Templates","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.13006","snapshot_observed_at":"2026-08-07T11:32:31.693251Z","title":"Systematic evaluation of llm-as-a-judge in llm alignment tasks: Explainable metrics and diverse prompt templates.arXiv preprint arXiv:2408.13006, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:31.693251Z"},"links":{"cited_paper":"/paper/2408.13006","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:4247f4446f44a253fbb35c151defaaa71b503acdb46e8d417a4f8f9f8e64f16e","observation_id":"b472d78c-92ea-4f16-8e06-1c1a56c278ee","resolution":{"observed_at":"2026-08-07T11:32:31.693251Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02743","last_updated":"2025-02-14T23:02:03Z","snapshot_observed_at":"2026-08-16T13:12:51.615126Z","submitted_at":"2024-10-03T17:55:13Z","title":"MA-RLHF: Reinforcement Learning from Human Feedback with Macro Actions","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02743","snapshot_observed_at":"2026-08-07T11:32:31.776781Z","title":"Ma-rlhf: Reinforcement learning from human feedback with macro actions.arXiv preprint arXiv:2410.02743, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:31.776781Z"},"links":{"cited_paper":"/paper/2410.02743","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:7285ba422f9a5cdba4793ba42ba2e1194722174094147f26305bf2cc5a058b46","observation_id":"71bc59d4-7290-4a87-b902-305f8be53e54","resolution":{"observed_at":"2026-08-07T11:32:31.776781Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-19T07:02:37.182654Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective"},"reference_resolution":{"displayed":43,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":35,"verified_exact":1,"verified_fuzzy":6},"total_outbound_references":43},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"thesis":"As of 21 August 2026, this Paper Citation Record lists 43 of 43 outbound references and 3 inbound Pith citation observations for arXiv:2506.02553."}