{"as_of":"2026-08-18T11:54:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:4a038b58dd2432ff9bccca39d5906857d7782f32a15f9808509069190900ba75","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":27,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":27,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-18T06:34:40.430872+00:00","state":"measured"},{"denominator":27,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":27,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T00:59:21.879268Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-08-16T12:54:50.426920Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-16T00:59:21.879268Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.02391","last_updated":"2025-05-05T06:26:00Z","snapshot_observed_at":"2026-08-17T18:37:06.996305Z","submitted_at":"2025-05-05T06:26:00Z","title":"Optimizing Chain-of-Thought Reasoners via Gradient Variance Minimization in Rejection Sampling and RL","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-16T00:59:21.879268Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2505.02391"},"observation_digest":"sha256:56cafda66c3ad1724dba185b986bc687e6f6f4e59396510b9092f91c75ae95b9","observation_id":"bac2e894-6082-431f-ac62-b315f956605a","resolution":{"observed_at":"2026-08-16T00:59:21.879268Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-08-16T12:54:50.426920Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-16T00:48:19.274791Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.02665","last_updated":"2025-05-08T05:27:18Z","snapshot_observed_at":"2026-08-16T00:50:49.069480Z","submitted_at":"2025-05-05T14:14:59Z","title":"A Survey of Slow Thinking-based Reasoning LLMs using Reinforced Learning and Inference-time Scaling Law","version":2},"reference_index":126,"source":"pdf_text","source_observed_at":"2026-08-16T00:48:19.274791Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2505.02665"},"observation_digest":"sha256:9ff5a5acb7693e92387789a4c53786b8474fddabc5c2901228165f78ddd71014","observation_id":"526399d3-1612-40f6-95c7-11537eb8752b","resolution":{"observed_at":"2026-08-16T00:48:19.274791Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-08-16T12:54:50.426920Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-15T23:13:33.639132Z","title":"Self- rewarding correction for mathematical reasoning.arXiv preprint arXiv:2502.19613, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.05315","last_updated":"2025-05-21T05:40:27Z","snapshot_observed_at":"2026-08-18T03:20:04.325807Z","submitted_at":"2025-05-08T15:01:06Z","title":"Scalable Chain of Thoughts via Elastic Reasoning","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-15T23:13:33.639132Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2505.05315"},"observation_digest":"sha256:1e090375689b7698043e35a5065416771be522741f1583013cfa1e1c824d094d","observation_id":"7840a340-42c9-4ada-a926-8534ee4d2064","resolution":{"observed_at":"2026-08-15T23:13:33.639132Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-08-16T12:54:50.426920Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-15T20:18:14.112520Z","title":"Xiong, H","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.13445","last_updated":"2025-05-19T17:59:31Z","snapshot_observed_at":"2026-08-17T21:57:34.658507Z","submitted_at":"2025-05-19T17:59:31Z","title":"Trust, But Verify: A Self-Verification Approach to Reinforcement Learning with Verifiable Rewards","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-15T20:18:14.112520Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2505.13445"},"observation_digest":"sha256:a8066f8302d54f6986d2a32325c267d04e140c778fc6e68b21d52ecf4cae3537","observation_id":"968ddbac-06f6-4cee-adb7-a6b99431d84a","resolution":{"observed_at":"2026-08-15T20:18:14.112520Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-08-16T12:54:50.426920Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-07T12:19:33.115481Z","title":"Xiong, H","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.24871","last_updated":"2025-06-05T05:13:46Z","snapshot_observed_at":"2026-08-14T12:50:25.203482Z","submitted_at":"2025-05-30T17:59:38Z","title":"MoDoMoDo: Multi-Domain Data Mixtures for Multimodal LLM Reinforcement Learning","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-07T12:19:33.115481Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2505.24871"},"observation_digest":"sha256:5a21225bc362c52a5e06eb1ccff91841374e1af7b55cbd947de8208e0a968482","observation_id":"33f319fc-9290-4edf-ac9a-6f53d1eb7b31","resolution":{"observed_at":"2026-08-07T12:19:33.115481Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-08-16T12:54:50.426920Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-07T05:51:30.740802Z","title":"Self-rewarding correction for mathematical reasoning.arXiv preprint arXiv:2502.19613,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.06923","last_updated":"2025-06-07T21:23:00Z","snapshot_observed_at":"2026-08-18T04:51:46.950390Z","submitted_at":"2025-06-07T21:23:00Z","title":"Boosting LLM Reasoning via Spontaneous Self-Correction","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T05:51:30.740802Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2506.06923"},"observation_digest":"sha256:5d474f986c4dc36d26bd5c12fbe330b3ee05fcf47b423312393369b447f87c19","observation_id":"a00c80d3-824f-4ab2-922a-582cb76fbd30","resolution":{"observed_at":"2026-08-07T05:51:30.740802Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-08-16T12:54:50.426920Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-07T05:09:45.657899Z","title":"Self-rewarding correction for mathematical reasoning.arXiv preprint arXiv:2502.19613, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.08745","last_updated":"2025-06-10T12:40:39Z","snapshot_observed_at":"2026-08-12T19:31:07.499130Z","submitted_at":"2025-06-10T12:40:39Z","title":"Consistent Paths Lead to Truth: Self-Rewarding Reinforcement Learning for LLM Reasoning","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T05:09:45.657899Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2506.08745"},"observation_digest":"sha256:5e56093759b30eb5ec53a1dc31f4b009a5d6b610c14e758cd3672b9857289343","observation_id":"3d8ec851-bcbe-4610-825e-f5be039ea645","resolution":{"observed_at":"2026-08-07T05:09:45.657899Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-08-16T12:54:50.426920Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-07T04:39:37.601699Z","title":"Self-rewarding correction for mathematical reasoning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.10406","last_updated":"2025-06-12T06:59:35Z","snapshot_observed_at":"2026-08-15T05:52:57.120730Z","submitted_at":"2025-06-12T06:59:35Z","title":"PAG: Multi-Turn Reinforced LLM Self-Correction with Policy as Generative Verifier","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T04:39:37.601699Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2506.10406"},"observation_digest":"sha256:34e94a3289f1e643682a9b7c383ca869808a8251f2885b421b6df187cb91f6f8","observation_id":"128b7cd0-6b9c-4786-9296-6e6a7a9fc255","resolution":{"observed_at":"2026-08-07T04:39:37.601699Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-08-16T12:54:50.426920Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-07T00:41:48.404875Z","title":"Xiong, H","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.12928","last_updated":"2025-06-15T17:59:47Z","snapshot_observed_at":"2026-08-14T20:05:17.894471Z","submitted_at":"2025-06-15T17:59:47Z","title":"Scaling Test-time Compute for LLM Agents","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T00:41:48.404875Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2506.12928"},"observation_digest":"sha256:ac6137f5815729ae8663d31a85a8b2fa5ae559353d196349332bc0a38ea29aec","observation_id":"97f62b28-d998-499f-aa58-9eb0923c1a72","resolution":{"observed_at":"2026-08-07T00:41:48.404875Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-08-16T12:54:50.426920Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":"2502.19613","doi":"10.48550/arxiv.2502.19613","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Self-rewarding correction for mathematical reasoning.arXiv preprint arXiv:2502.19613","venue":"ArXiv.org","work_id":"1ceb5bfe-8d1e-431e-b67a-458ea30e168e","year":2025},"citing_paper":{"arxiv_id":"2509.03403","last_updated":"2026-05-15T21:29:46Z","snapshot_observed_at":"2026-08-15T01:33:58.861430Z","submitted_at":"2025-09-03T15:28:51Z","title":"Beyond Correctness: Harmonizing Process and Outcome Rewards through RL Training","version":2},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-05-21T22:38:57.833414Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2509.03403"},"observation_digest":"sha256:8f84e4fc783ec14f203c815ac943ac2b65e136d539971f3ec3450e71014cc5df","observation_id":"cc1bd704-6917-4eda-a7cf-32a282e1a413","resolution":{"observed_at":"2026-05-21T22:40:43.171530Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-08-16T12:54:50.426920Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":"2502.19613","doi":"10.48550/arxiv.2502.19613","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Self-rewarding correction for mathematical reasoning.arXiv preprint arXiv:2502.19613","venue":"ArXiv.org","work_id":"1ceb5bfe-8d1e-431e-b67a-458ea30e168e","year":2025},"citing_paper":{"arxiv_id":"2510.07962","last_updated":"2026-05-21T01:02:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-09T08:55:12Z","title":"LightReasoner: Can Small Language Models Teach Large Language Models Reasoning?","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-22T12:59:31.011283Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2510.07962"},"observation_digest":"sha256:85eda2e928abbd490347861d9d92d6d378d7ae490a8328d2bad8f6969e9fd910","observation_id":"06743f5c-b56f-4ab0-a5ae-fcede9ccce86","resolution":{"observed_at":"2026-05-22T13:01:34.019544Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-08-16T12:54:50.426920Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-04T10:44:31.460549Z","title":"Self-rewarding correction for mathematical reasoning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2510.08977","last_updated":"2026-06-02T08:25:17Z","snapshot_observed_at":"2026-08-16T06:07:54.370427Z","submitted_at":"2025-10-10T03:38:17Z","title":"Breaking the Self-Confirming Loop: Diagnosing and Mitigating Systemic Reward Bias in Self-Rewarding RL","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-04T10:44:31.460549Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2510.08977"},"observation_digest":"sha256:9403736b988dd42f9e2b726bfcf3585dcaaf2f6e3566e0d874ea913e9238fbc3","observation_id":"5830c2c4-5a28-44d0-9fea-33295906121a","resolution":{"observed_at":"2026-08-04T10:44:31.460549Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-08-16T12:54:50.426920Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-03T05:14:21.533431Z","title":"Self-rewarding correction for mathematical reasoning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.02979","last_updated":"2026-05-24T21:03:36Z","snapshot_observed_at":"2026-08-12T09:46:52.523380Z","submitted_at":"2026-02-03T01:38:53Z","title":"CPMobius: Iterative Coach-Player Reasoning for Data-Free Reinforcement Learning","version":3},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-03T05:14:21.533431Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2602.02979"},"observation_digest":"sha256:954566f20203be9089dc32527db514f40b8f5cd11a04a07e0bd8b0dde5729b2b","observation_id":"b896a909-8106-4d06-ac64-1436ef0624b9","resolution":{"observed_at":"2026-08-03T05:14:21.533431Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-08-16T12:54:50.426920Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":"2502.19613","doi":"10.48550/arxiv.2502.19613","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Self-rewarding correction for mathematical reasoning.arXiv preprint arXiv:2502.19613","venue":"ArXiv.org","work_id":"1ceb5bfe-8d1e-431e-b67a-458ea30e168e","year":2025},"citing_paper":{"arxiv_id":"2602.07832","last_updated":"2026-07-03T17:35:46Z","snapshot_observed_at":"2026-08-16T06:39:12.822315Z","submitted_at":"2026-02-08T05:47:27Z","title":"rePIRL: Learn PRM with Inverse RL for LLM Reasoning","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-21T13:13:13.293921Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2602.07832"},"observation_digest":"sha256:294ab3ac8fb2daab6b5021e319f1ce8cecae2087766c52f4039c5e62623b7336","observation_id":"81f621f4-3604-428e-8cc2-e0bf2c72abc3","resolution":{"observed_at":"2026-05-21T13:14:10.976923Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-08-16T12:54:50.426920Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-03T03:33:45.150810Z","title":"Self-rewarding correction for mathematical reasoning.arXiv preprint arXiv:2502.19613, 2025a","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.07832","last_updated":"2026-07-03T17:35:46Z","snapshot_observed_at":"2026-08-16T06:39:12.822315Z","submitted_at":"2026-02-08T05:47:27Z","title":"rePIRL: Learn PRM with Inverse RL for LLM Reasoning","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-03T03:33:45.150810Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2602.07832"},"observation_digest":"sha256:750379dc072def825b570815a900828f957fc6c334c19cbc7488dddedca1be85","observation_id":"dbfc3b72-9643-43e4-85a6-9b411c13cfd7","resolution":{"observed_at":"2026-08-03T03:33:45.150810Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-08-16T12:54:50.426920Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":"2502.19613","doi":"10.48550/arxiv.2502.19613","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Self-rewarding correction for mathematical reasoning.arXiv preprint arXiv:2502.19613","venue":"ArXiv.org","work_id":"1ceb5bfe-8d1e-431e-b67a-458ea30e168e","year":2025},"citing_paper":{"arxiv_id":"2604.03993","last_updated":"2026-04-05T06:30:50Z","snapshot_observed_at":"2026-07-06T22:53:01.835843Z","submitted_at":"2026-04-05T06:30:50Z","title":"Can LLMs Learn to Reason Robustly under Noisy Supervision?","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-13T16:58:42.129870Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2604.03993"},"observation_digest":"sha256:bdd28a92dbba41bcfbaa6b5c7b6166936b9086bcaef1d789b909c57b0e21b9c7","observation_id":"45e07ccf-0df1-4722-b143-7e48f81aada3","resolution":{"observed_at":"2026-05-13T17:08:01.264453Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-08-16T12:54:50.426920Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":"2502.19613","doi":"10.48550/arxiv.2502.19613","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Self-rewarding correction for mathematical reasoning.arXiv preprint arXiv:2502.19613","venue":"ArXiv.org","work_id":"1ceb5bfe-8d1e-431e-b67a-458ea30e168e","year":2025},"citing_paper":{"arxiv_id":"2605.05226","last_updated":"2026-05-23T13:20:07Z","snapshot_observed_at":"2026-08-11T12:30:16.051529Z","submitted_at":"2026-04-19T10:33:19Z","title":"Internalizing Outcome Supervision into Process Supervision: A New Paradigm for Reinforcement Learning for Reasoning","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-10T06:13:09.898530Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2605.05226"},"observation_digest":"sha256:ea21a1bd0158bc8ae377097f743bc18ac2ca492d2edd0218193f920a75b12c14","observation_id":"6c7cb4b3-0047-45bd-a6ed-f07c5cd35174","resolution":{"observed_at":"2026-05-10T06:26:27.829461Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-08-16T12:54:50.426920Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":"2502.19613","doi":"10.48550/arxiv.2502.19613","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Self-rewarding correction for mathematical reasoning.arXiv preprint arXiv:2502.19613","venue":"ArXiv.org","work_id":"1ceb5bfe-8d1e-431e-b67a-458ea30e168e","year":2025},"citing_paper":{"arxiv_id":"2605.16299","last_updated":"2026-05-21T14:01:21Z","snapshot_observed_at":"2026-08-15T18:25:33.406812Z","submitted_at":"2026-04-17T07:20:01Z","title":"ACE: Self-Evolving LLM Coding Framework via Adversarial Unit Test Generation and Preference Optimization","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-21T00:55:40.784295Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2605.16299"},"observation_digest":"sha256:4bdf042382fd030574473213d2d6d0d264d7aec9011c5173237d850e16d77921","observation_id":"cf61d227-58ea-4435-9385-6261c2420a7c","resolution":{"observed_at":"2026-05-21T00:59:19.336113Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-08-16T12:54:50.426920Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":"2502.19613","doi":"10.48550/arxiv.2502.19613","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Self-rewarding correction for mathematical reasoning.arXiv preprint arXiv:2502.19613","venue":"ArXiv.org","work_id":"1ceb5bfe-8d1e-431e-b67a-458ea30e168e","year":2025},"citing_paper":{"arxiv_id":"2605.16299","last_updated":"2026-05-21T14:01:21Z","snapshot_observed_at":"2026-08-15T18:25:33.406812Z","submitted_at":"2026-04-17T07:20:01Z","title":"ACE: Self-Evolving LLM Coding Framework via Adversarial Unit Test Generation and Preference Optimization","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-22T10:14:00.478000Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2605.16299"},"observation_digest":"sha256:252bb47d794494632fea2b984c4ffe00aa410f3da175cee00ed2cc446573fd94","observation_id":"f414539e-896d-4e51-bb05-6e65dc61007e","resolution":{"observed_at":"2026-05-22T10:14:47.165809Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-08-16T12:54:50.426920Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":"2502.19613","doi":"10.48550/arxiv.2502.19613","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Self-rewarding correction for mathematical reasoning.arXiv preprint arXiv:2502.19613","venue":"ArXiv.org","work_id":"1ceb5bfe-8d1e-431e-b67a-458ea30e168e","year":2025},"citing_paper":{"arxiv_id":"2605.24613","last_updated":"2026-05-23T14:51:13Z","snapshot_observed_at":"2026-08-15T17:07:09.421006Z","submitted_at":"2026-05-23T14:51:13Z","title":"Guarded Repair for Harm-Aware Post-hoc Replacement of LLM Mathematical Reasoning","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-06-30T13:38:48.617204Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2605.24613"},"observation_digest":"sha256:da84d49623778b5d799ac6089c0fd882ce2c6be95f566efc276d43bb77c5b86f","observation_id":"0dd7aacd-e04f-4881-a417-eea4ab73c190","resolution":{"observed_at":"2026-06-30T13:44:40.809593Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-08-16T12:54:50.426920Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":"2502.19613","doi":"10.48550/arxiv.2502.19613","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Self-rewarding correction for mathematical reasoning.arXiv preprint arXiv:2502.19613","venue":"ArXiv.org","work_id":"1ceb5bfe-8d1e-431e-b67a-458ea30e168e","year":2025},"citing_paper":{"arxiv_id":"2606.01249","last_updated":"2026-06-17T04:44:10Z","snapshot_observed_at":"2026-08-12T03:28:30.781632Z","submitted_at":"2026-05-31T14:04:51Z","title":"Trust Region On-Policy Distillation","version":3},"reference_index":162,"source":"arxiv_source","source_observed_at":"2026-06-28T17:38:50.313305Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2606.01249"},"observation_digest":"sha256:8db2c163b66cd37dcc57f6750923da749f2fc995ad9040d7af899ebfb12ceb9b","observation_id":"03fe794f-2c38-44d0-961d-2ae2f83409a2","resolution":{"observed_at":"2026-07-01T20:56:13.435749Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-08-16T12:54:50.426920Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":"2502.19613","doi":"10.48550/arxiv.2502.19613","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Self-rewarding correction for mathematical reasoning.arXiv preprint arXiv:2502.19613","venue":"ArXiv.org","work_id":"1ceb5bfe-8d1e-431e-b67a-458ea30e168e","year":2025},"citing_paper":{"arxiv_id":"2606.11470","last_updated":"2026-08-10T19:13:02Z","snapshot_observed_at":"2026-08-14T23:09:31.705348Z","submitted_at":"2026-06-09T21:59:37Z","title":"The Periodic Table of LLM Reasoning: A Structured Survey of Reasoning Paradigms, Methods, and Failure Modes","version":1},"reference_index":274,"source":"arxiv_source","source_observed_at":"2026-06-27T12:59:51.091008Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2606.11470"},"observation_digest":"sha256:aa6f80a2c60d8d622e919a996f52d9d06aef0c0335aab6887b35278715792ffc","observation_id":"c9f7a4fa-0632-4437-b449-b34408d22a14","resolution":{"observed_at":"2026-06-27T13:00:55.970436Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-08-16T12:54:50.426920Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":"2502.19613","doi":"10.48550/arxiv.2502.19613","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Self-rewarding correction for mathematical reasoning.arXiv preprint arXiv:2502.19613","venue":"ArXiv.org","work_id":"1ceb5bfe-8d1e-431e-b67a-458ea30e168e","year":2025},"citing_paper":{"arxiv_id":"2606.13316","last_updated":"2026-07-31T08:57:20Z","snapshot_observed_at":"2026-08-13T10:34:20.953214Z","submitted_at":"2026-06-11T13:10:48Z","title":"ReSum: Synergizing LLM Reasoning and Summarization with Reinforcement Learning","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-06-27T06:39:34.199607Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2606.13316"},"observation_digest":"sha256:56f971c31ca82b6cab1201e81c492ae3ba01b93a6cb0e2e95a4e7188c79500da","observation_id":"86fff995-f14c-4448-b088-6cff9d7586e5","resolution":{"observed_at":"2026-07-03T15:08:33.388470Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-08-16T12:54:50.426920Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-03T02:12:24.010055Z","title":"Xiong, H","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.13316","last_updated":"2026-07-31T08:57:20Z","snapshot_observed_at":"2026-08-13T10:34:20.953214Z","submitted_at":"2026-06-11T13:10:48Z","title":"ReSum: Synergizing LLM Reasoning and Summarization with Reinforcement Learning","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-03T02:12:24.010055Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2606.13316"},"observation_digest":"sha256:346ea5903cbb40e627ce7e34aa5e08c0815ef4ff21c0029e9bc1753752bd10ab","observation_id":"deeadaf6-7404-490b-a02c-e0291d1527f5","resolution":{"observed_at":"2026-08-03T02:12:24.010055Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-08-16T12:54:50.426920Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":"2502.19613","doi":"10.48550/arxiv.2502.19613","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Self-rewarding correction for mathematical reasoning.arXiv preprint arXiv:2502.19613","venue":"ArXiv.org","work_id":"1ceb5bfe-8d1e-431e-b67a-458ea30e168e","year":2025},"citing_paper":{"arxiv_id":"2606.21943","last_updated":"2026-06-20T08:20:41Z","snapshot_observed_at":"2026-08-15T21:55:25.625601Z","submitted_at":"2026-06-20T08:20:41Z","title":"Modularized Reinforcement Learning on LLMs: From MDP Creation to Exploration and Learning","version":1},"reference_index":238,"source":"pdf_text","source_observed_at":"2026-06-26T12:15:08.304150Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2606.21943"},"observation_digest":"sha256:3f0c465492b41985e71172a9da14225e1ca6169b96c52d236033ac77822b3028","observation_id":"2cf7f342-4619-40d3-8df7-e5cc461b49d7","resolution":{"observed_at":"2026-07-04T07:59:40.132829Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-08-16T12:54:50.426920Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":"2502.19613","doi":"10.48550/arxiv.2502.19613","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Self-rewarding correction for mathematical reasoning.arXiv preprint arXiv:2502.19613","venue":"ArXiv.org","work_id":"1ceb5bfe-8d1e-431e-b67a-458ea30e168e","year":2025},"citing_paper":{"arxiv_id":"2606.28186","last_updated":"2026-08-07T19:15:46Z","snapshot_observed_at":"2026-08-16T05:45:39.781887Z","submitted_at":"2026-06-26T15:32:17Z","title":"Cognitive Episodes in LLM Reasoning Traces Enable Interpretable Human Item Difficulty Prediction","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-29T03:58:32.372896Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2606.28186"},"observation_digest":"sha256:3219217b868c7573e4def3b4f448ddfab64aa6e086bfd8b6b0faa313e2be696c","observation_id":"af7e3537-efcf-4951-ba7a-d7583bc4b8f7","resolution":{"observed_at":"2026-07-01T17:15:51.588940Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-08-16T12:54:50.426920Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-07-14T17:12:24.565155Z","title":"Self-rewarding correction for mathematical reasoning.arXiv preprint arXiv:2502.19613,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.28186","last_updated":"2026-08-07T19:15:46Z","snapshot_observed_at":"2026-08-16T05:45:39.781887Z","submitted_at":"2026-06-26T15:32:17Z","title":"Cognitive Episodes in LLM Reasoning Traces Enable Interpretable Human Item Difficulty Prediction","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-07-14T17:12:24.565155Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2606.28186"},"observation_digest":"sha256:60f2b3e1a065e51d19f80538b8da7891c9fb170a745d589a078de7f244306a3d","observation_id":"8428ec32-f496-4353-9265-0eb5a8b61781","resolution":{"observed_at":"2026-07-14T17:12:24.565155Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2502.19613/citation-record","integrity":"/paper/2502.19613/integrity","json":"/paper/2502.19613/citation-record.json","paper":"/paper/2502.19613"},"outbound":[],"paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-16T12:54:50.426920Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 27 inbound Pith citation observations for arXiv:2502.19613."}