{"as_of":"2026-08-18T00:23:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:3d6d7da886f0bef637767277349cd854dc48b3650539c43d2e37be8b0d0972d5","coverage":[{"denominator":34,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":34,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-10T18:52:52.969408Z","state":"measured"},{"denominator":35,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":35,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-17T06:30:58.91139+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-14T07:36:47.336670Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.02909","snapshot_observed_at":"2026-07-14T07:36:47.336670Z","title":"Dorner, Robin Staab, and Martin Vechev","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.11022","last_updated":"2026-07-13T02:41:16Z","snapshot_observed_at":"2026-08-14T14:34:39.089083Z","submitted_at":"2026-07-13T02:41:16Z","title":"When the Reward Suite Is Leaky: A Preregistered Causal Contrast of Natural Verifier False Positives in RLVR","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-14T07:36:47.336670Z"},"links":{"cited_paper":"/paper/2605.02909","citing_paper":"/paper/2607.11022"},"observation_digest":"sha256:edbedfa851cb380ad672990f6320dd033486bc079d26ebd1366a5f83b465bcfa","observation_id":"59408003-0a79-4dac-9265-e33eae1e212f","resolution":{"observed_at":"2026-07-14T07:36:47.336670Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2605.02909/citation-record","integrity":"/paper/2605.02909/integrity","json":"/paper/2605.02909/citation-record.json","paper":"/paper/2605.02909"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2508.06471","last_updated":"2025-08-08T17:21:06Z","snapshot_observed_at":"2026-08-11T03:34:09.767397Z","submitted_at":"2025-08-08T17:21:06Z","title":"GLM-4.5: Agentic, Reasoning, and Coding (ARC) Foundation Models","version":1},"cited_work":{"arxiv_id":"2508.06471","doi":"10.48550/arxiv.2508.06471","metadata_source":"pith","pith_arxiv_id":"2508.06471","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GLM-4.5: Agentic, Reasoning, and Coding (ARC) Foundation Models","venue":"cs.CL","work_id":"5bb4e5d7-985e-431b-bb0d-75576cdc2950","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2508.06471","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:ae04a8b84004e742274e4419e0f778ef094142314d2fcf33edf8ed60fcb3128b","observation_id":"76519476-d543-4cbf-a667-29ebfd6c96cb","resolution":{"observed_at":"2026-05-11T17:50:08.654972Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-05-23T05:23:01.474969+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T05:23:01.474969+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":"2402.03300","doi":"10.1016/0004-3702(73)90011-8","metadata_source":"pith","pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","venue":"cs.CL","work_id":"c5006563-f3ec-438a-9e35-b7b484f34828","year":2024},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:39d5912f243888a9abd2761fb16c4862a35e191fcea4a6706ba3f74e5c8f3601","observation_id":"125e9404-bc66-4dca-a54f-220ea5b242bb","resolution":{"observed_at":"2026-05-10T23:45:53.096134Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.17746","last_updated":"2025-10-03T01:55:55Z","snapshot_observed_at":"2026-08-12T14:23:22.826603Z","submitted_at":"2025-07-23T17:57:55Z","title":"Rubrics as Rewards: Reinforcement Learning Beyond Verifiable Domains","version":2},"cited_work":{"arxiv_id":"2507.17746","doi":"10.48550/arxiv.2507.17746","metadata_source":"pith","pith_arxiv_id":"2507.17746","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Rubrics as Rewards: Reinforcement Learning Beyond Verifiable Domains","venue":"cs.LG","work_id":"805a846c-dae9-4375-abd8-a86dc6934496","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2507.17746","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:33bcf7eec8030598573ef523aa41568d010c2d1a03bfc99eb4798b11505b0a16","observation_id":"3b43911a-7a37-4e8b-b14f-9467ebd4b44b","resolution":{"observed_at":"2026-05-13T06:07:56.884714Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-05-23T21:53:02.879284+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T21:53:02.879284+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2601.04411","doi":"10.48550/arxiv.2601.04411","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Rate or fate? rlv r: Reinforcement learning with verifiable noisy rewards","venue":"arXiv (Cornell University)","work_id":"a97bd8c0-0af9-40f1-ac54-e7b6759bf1f4","year":2026},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:fd7d9140919fb80b963bcd5a2ebbd81562ca7e7efa080f474e76f3a0a21526cf","observation_id":"5e6f45d2-ab36-4b79-8384-7d4ef0c296ad","resolution":{"observed_at":"2026-05-10T23:45:53.041417Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.00915","last_updated":"2026-05-22T12:16:11Z","snapshot_observed_at":"2026-08-15T01:19:02.750743Z","submitted_at":"2025-10-01T13:56:44Z","title":"Reinforcement Learning with Verifiable yet Noisy Rewards under Imperfect Verifiers","version":4},"cited_work":{"arxiv_id":"2510.00915","doi":null,"metadata_source":"pith","pith_arxiv_id":"2510.00915","snapshot_observed_at":"2026-07-04T17:09:58.372131Z","title":"Reinforcement Learning with Verifiable yet Noisy Rewards under Imperfect Verifiers","venue":"cs.LG","work_id":"12cfcad6-e768-4ecf-9455-0e5bdb645f9a","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2510.00915","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:070c3aa90104dc8707ef2f49b047322f72c438491fbd14e7d0fb8a7405542e94","observation_id":"8e146cf6-da76-484f-8e82-84205622ea8f","resolution":{"observed_at":"2026-05-25T03:01:06.910934Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.22653","last_updated":"2025-05-28T17:59:03Z","snapshot_observed_at":"2026-08-16T11:18:20.043444Z","submitted_at":"2025-05-28T17:59:03Z","title":"The Climb Carves Wisdom Deeper Than the Summit: On the Noisy Rewards in Learning to Reason","version":1},"cited_work":{"arxiv_id":"2505.22653","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.22653","snapshot_observed_at":"2026-07-02T06:06:40.819064Z","title":"The climb carves wisdom deeper than the summit: On the noisy rewards in learning to reason","venue":null,"work_id":"6e345e0c-7c39-4e94-af60-56d7643c5704","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2505.22653","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:7aa05b0dc944f70d83324b90f197280df9e66e38c728363b7838fb096e0fcf33","observation_id":"108954cf-c305-4ec2-8327-6849d809a05d","resolution":{"observed_at":"2026-05-10T23:45:53.070918Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.22203","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-02T06:06:40.751284Z","title":"Pitfalls of rule- and model-based verifiers–a case study on mathematical reasoning","venue":null,"work_id":"606127e0-cfd7-4548-893d-87fb5c3ea69e","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:ce043024a90f717666acf1ede5e9c93e1281403d0d49356d5e2d48138b96ebd8","observation_id":"fada4551-957b-4858-8317-bc22f9733c1f","resolution":{"observed_at":"2026-05-10T23:45:52.968599Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.16400","last_updated":"2025-06-05T17:59:12Z","snapshot_observed_at":"2026-08-17T12:06:20.372156Z","submitted_at":"2025-05-22T08:50:47Z","title":"AceReason-Nemotron: Advancing Math and Code Reasoning through Reinforcement Learning","version":3},"cited_work":{"arxiv_id":"2505.16400","doi":"10.48550/arxiv.2505.16400","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.16400","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Acereason-nemotron: Advancing math and code reasoning through reinforcement learning","venue":"ArXiv.org","work_id":"428ad314-c120-41da-9db7-b8bc1918fffb","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2505.16400","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:2d9367aa11432fc663d9b3ef893ad1655cfc690f9d13d61815f5e79c380bf09c","observation_id":"edc3109a-db52-4c8f-b539-c362704567e8","resolution":{"observed_at":"2026-05-10T23:45:53.101099Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.08794","last_updated":"2026-06-11T04:21:36Z","snapshot_observed_at":"2026-08-17T18:32:55.983199Z","submitted_at":"2025-07-11T17:55:22Z","title":"One Token to Fool LLM-as-a-Judge","version":3},"cited_work":{"arxiv_id":"2507.08794","doi":"10.48550/arxiv.2507.08794","metadata_source":"pith","pith_arxiv_id":"2507.08794","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"One token to fool llm-as-a-judge","venue":"cs.LG","work_id":"f77305eb-5f89-4afb-84f3-51d864f3f3fa","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2507.08794","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:3c56d98e83626e9307d5445ef4e830dbe76414e98ceb3222fd8c15f38f2230e5","observation_id":"12f775e4-d555-420d-91f6-8801f592afdd","resolution":{"observed_at":"2026-06-12T02:08:19.458599Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.24760","doi":"10.48550/arxiv.2505.24760","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reasoning gym: Reasoning environments for reinforcement learning with verifiable rewards.arXiv preprint arXiv:2505.24760","venue":"ArXiv.org","work_id":"dfc22a2c-1b8e-4318-a044-e142c0608571","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:2d6687722f9abf51d5171eaae6fe82f50d83b26e646ef3cef3309ec6bc2c587d","observation_id":"310b2449-9f46-43bb-87e0-94e585c3d0f2","resolution":{"observed_at":"2026-05-10T23:45:53.036235Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-08-15T12:33:55.451951Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2501.12948","doi":"10.1016/j.artmed.2024.103001","metadata_source":"pith","pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","venue":"cs.CL","work_id":"e6b75ad5-2877-4168-97c8-710407094d20","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:5904b652b3f14d10ae05736875f2a48e3235ea35a7f7062546114c21b5bd247f","observation_id":"694ad195-4ca0-4d2e-a90c-7b25eaa354b6","resolution":{"observed_at":"2026-05-10T23:45:53.128053Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.20534","last_updated":"2026-02-03T04:57:00Z","snapshot_observed_at":"2026-08-16T14:37:33.548231Z","submitted_at":"2025-07-28T05:35:43Z","title":"Kimi K2: Open Agentic Intelligence","version":2},"cited_work":{"arxiv_id":"2507.20534","doi":"10.1145/3448609","metadata_source":"pith","pith_arxiv_id":"2507.20534","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Kimi K2: Open Agentic Intelligence","venue":"cs.LG","work_id":"7f18284c-12d3-4137-bea1-1da97e8cf3c1","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2507.20534","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:f64c8afc319b7baec04f3e5dbe6bf517ea362bc6dc059f3b52fc601b3d2aa411","observation_id":"302b7624-4922-40a2-a517-468eae920624","resolution":{"observed_at":"2026-05-10T23:45:53.056780Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-05-25T01:23:16.170083+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-25T01:23:16.170083+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":"2505.09388","doi":"10.1016/j.aiopen.2022.12","metadata_source":"pith","pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Qwen3 Technical Report","venue":"cs.CL","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:d47a10df20e538300e175717a5d1b92979f879945f5dc84b683df6fb8c6499d9","observation_id":"c5ef677f-14c1-4d1a-aa23-1eb46de404c0","resolution":{"observed_at":"2026-05-10T23:45:52.995730Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.03613","last_updated":"2025-08-05T16:28:22Z","snapshot_observed_at":"2026-08-11T11:12:15.356883Z","submitted_at":"2025-08-05T16:28:22Z","title":"Goedel-Prover-V2: Scaling Formal Theorem Proving with Scaffolded Data Synthesis and Self-Correction","version":1},"cited_work":{"arxiv_id":"2508.03613","doi":"10.48550/arxiv.2508.03613","metadata_source":"pith","pith_arxiv_id":"2508.03613","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Goedel-Prover-V2: Scaling Formal Theorem Proving with Scaffolded Data Synthesis and Self-Correction","venue":"cs.LG","work_id":"cfdb69bb-3c0b-41d3-bd34-3167a6931bb2","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2508.03613","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:cb9f869885f85a188578a9c7f6303a4a5c2d37311e64ef38653fa7100dae6e1e","observation_id":"d640b613-6f87-48a8-9aca-21fcbc27a503","resolution":{"observed_at":"2026-05-21T06:53:10.886894Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-07-12T04:49:11.143969+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T04:49:11.143969+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"RLTF: reinforcement learning from unit test feedback","venue":null,"work_id":"7c0127cf-c687-4a6d-8643-e00e5d5ffe2b","year":2023},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:55eeec136bca7710972d18d5994d49c08bc5ce58c3c099bd25c8572092adda5d","observation_id":"0406a932-dfb8-4e0c-a147-923635c8482c","resolution":{"observed_at":"2026-05-16T13:22:55.861396Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.18449","last_updated":"2025-12-01T00:16:59Z","snapshot_observed_at":"2026-08-13T07:37:18.494967Z","submitted_at":"2025-02-25T18:45:04Z","title":"SWE-RL: Advancing LLM Reasoning via Reinforcement Learning on Open Software Evolution","version":2},"cited_work":{"arxiv_id":"2502.18449","doi":"10.48550/arxiv.2502.18449","metadata_source":"pith","pith_arxiv_id":"2502.18449","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-RL: Advancing LLM Reasoning via Reinforcement Learning on Open Software Evolution","venue":"cs.SE","work_id":"4b93fb93-87c9-40fe-84d0-d7ecb4e11bed","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2502.18449","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:98058bb9bfaca0887c69518d0a03f23b97707b7d229862cd8fb8265ef1add6aa","observation_id":"1af04395-3132-40a2-90c1-4994093e95d2","resolution":{"observed_at":"2026-05-15T10:27:56.991478Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.14476","last_updated":"2025-05-20T01:37:34Z","snapshot_observed_at":"2026-08-02T01:40:54.187278Z","submitted_at":"2025-03-18T17:49:06Z","title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","version":2},"cited_work":{"arxiv_id":"2503.14476","doi":"10.48550/arxiv.2503.14476","metadata_source":"pith","pith_arxiv_id":"2503.14476","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","venue":"cs.LG","work_id":"64019d00-0b11-4bbd-b173-b46c8fad0157","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2503.14476","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:32d4db74d062aa9849f9ab169e631f49b73ebe58dc7c31ff09f1c09ab7743fa0","observation_id":"f2a41475-0f73-4145-9baa-caaa43c43256","resolution":{"observed_at":"2026-05-10T23:45:52.981204Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-08T16:08:20.547492+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T16:08:20.547492+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.20783","last_updated":"2025-10-06T09:30:03Z","snapshot_observed_at":"2026-08-13T12:34:54.476684Z","submitted_at":"2025-03-26T17:59:14Z","title":"Understanding R1-Zero-Like Training: A Critical Perspective","version":2},"cited_work":{"arxiv_id":"2503.20783","doi":"10.48550/arxiv.2503.20783","metadata_source":"pith","pith_arxiv_id":"2503.20783","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding R1-Zero-Like Training: A Critical Perspective","venue":"cs.LG","work_id":"ec354f3b-9484-4a0c-94c8-92d4d0260835","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2503.20783","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:3bc0db3771eaa104bcd778cc3b5f16eadcccbb77d111e2dbe64b337b49c2d004","observation_id":"5ccaaeda-4316-4973-b9f5-36a828d9efa7","resolution":{"observed_at":"2026-05-10T23:45:52.974750Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-05-24T09:23:05.84445+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-24T09:23:05.84445+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2511.20347","last_updated":"2025-12-01T12:02:46Z","snapshot_observed_at":"2026-08-12T16:05:03.602321Z","submitted_at":"2025-11-25T14:25:19Z","title":"Soft Adaptive Policy Optimization","version":2},"cited_work":{"arxiv_id":"2511.20347","doi":"10.48550/arxiv.2511.20347","metadata_source":"pith","pith_arxiv_id":"2511.20347","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Soft Adaptive Policy Optimization","venue":"cs.LG","work_id":"2d2c49f8-7feb-48c2-af7e-025b94c5d4f9","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2511.20347","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:a1fd75ef79d92fc44a23d21b89ed8814ae3138b2a214e21c80620df0d6715f43","observation_id":"893a8615-78e6-4c5a-84c2-66ad6ed087ac","resolution":{"observed_at":"2026-05-15T07:14:32.897338Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Is your code generated by chatgpt really correct? rigorous evaluation of large language models for code generation","venue":null,"work_id":"2b7e051f-7a24-452d-b3bb-0064170449bc","year":2023},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:f6a2cec09e5b70e83d397c2e76cc40eb5555eb2ef7dc1e47d11295eca862a5aa","observation_id":"43a860c9-3c65-400b-85f4-4d3c0c6a5a5e","resolution":{"observed_at":"2026-05-16T13:22:55.866308Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14625","last_updated":"2025-05-22T17:49:50Z","snapshot_observed_at":"2026-08-14T03:49:10.492607Z","submitted_at":"2025-05-20T17:16:44Z","title":"TinyV: Reducing False Negatives in Verification Improves RL for LLM Reasoning","version":2},"cited_work":{"arxiv_id":"2505.14625","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.14625","snapshot_observed_at":"2026-07-02T02:16:26.183764Z","title":"Tinyv: Reducing false negatives in verification improves rl for llm reasoning","venue":null,"work_id":"851de516-697c-423c-8c2b-4f5fbdf5ac13","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2505.14625","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:6e435a0433968f6e2cc17d63cd9d0ad7dd19426d855c2d32969ad1b617680d58","observation_id":"42f34f68-8adb-4ba9-baf8-69569e51f279","resolution":{"observed_at":"2026-05-10T23:45:53.117674Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Is llm-as-a-judge robust? investigating universal adversarial attacks on zero-shot llm assessment","venue":null,"work_id":"138ade5b-7e26-4ede-bb04-5f663b09fd96","year":2024},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:0758a45c69d2168ed4ecf016fdaaab75e2eb75759dec65e07a35b810070bea22","observation_id":"710d6935-db0d-49c8-9d07-d62b7b60cb34","resolution":{"observed_at":"2026-05-16T13:22:55.876070Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.17995","last_updated":"2026-04-14T03:25:43Z","snapshot_observed_at":"2026-08-16T12:41:51.570336Z","submitted_at":"2025-09-22T16:36:56Z","title":"Variation in Verification: Understanding Verification Dynamics in Large Language Models","version":2},"cited_work":{"arxiv_id":"2509.17995","doi":null,"metadata_source":"pith","pith_arxiv_id":"2509.17995","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Variation in Verification: Understanding Verification Dynamics in Large Language Models","venue":"cs.CL","work_id":"f614fdf8-06a5-45a4-a7b0-d2ded78a7289","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2509.17995","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:396e9bd6f6fd749609f816dd5fa98b963ebd6e3d339afeb8528f87404c2cc1fd","observation_id":"f37d6420-c18b-4d4b-98c2-91da1bae9668","resolution":{"observed_at":"2026-05-10T23:45:53.052014Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Verifybench: A systematic benchmark for evaluating reasoning verifiers across domains","venue":null,"work_id":"6b608204-4555-4335-8e2d-d9d5cfa98293","year":2026},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:2c6e783e6bc0ff52c752f3e2f33721fceeadf3ae769ee9b3ddc46d5adc2628ae","observation_id":"73724eae-10b4-4a4f-a4a0-3e9924140e7e","resolution":{"observed_at":"2026-05-16T13:22:55.868976Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.15801","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Pure Python","venue":null,"work_id":"65a36a53-2af5-4e5a-82ff-806809a22ae4","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:1f2f79f403e0f51a50d9945075b27566a508245e8c366e003a8cc91e9247dc98","observation_id":"c42e22a1-a308-4f1a-976e-c9143320e6d0","resolution":{"observed_at":"2026-05-10T23:45:53.008410Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2603.16140","last_updated":"2026-07-29T22:29:54Z","snapshot_observed_at":"2026-08-02T23:16:34.332808Z","submitted_at":"2026-03-17T05:48:32Z","title":"Noisy Data is Destructive to Reinforcement Learning with Verifiable Rewards","version":2},"cited_work":{"arxiv_id":"2603.16140","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2603.16140","snapshot_observed_at":"2026-07-31T02:03:17.773516Z","title":"Noisy data is destructive to reinforcement learning with verifiable rewards","venue":null,"work_id":"b0966eb6-27a5-4d71-9205-f8df1710d940","year":2026},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2603.16140","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:cc980c6b226563d88a0fae86a83f5d1409cf63218251e8a44983f2494ee37e30","observation_id":"fc0d0878-2e23-48f6-add8-555fa72e8844","resolution":{"observed_at":"2026-07-31T02:03:17.773516Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Russell, and Anca D","venue":null,"work_id":"a4606a2d-cb02-4892-836f-30997132bfc4","year":2017},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:330d1ffe70bcbc2e271dff3943554e52d6cf763cc11ba6f3c3c5f50dc135ea49","observation_id":"c9795332-5524-49f9-8460-35aa937a5903","resolution":{"observed_at":"2026-05-16T13:22:55.855833Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2209.13085","last_updated":"2025-03-05T21:08:30Z","snapshot_observed_at":"2026-08-16T16:29:40.254423Z","submitted_at":"2022-09-27T00:32:44Z","title":"Defining and Characterizing Reward Hacking","version":2},"cited_work":{"arxiv_id":"2209.13085","doi":"10.48550/arxiv.2209.13085","metadata_source":"pith","pith_arxiv_id":"2209.13085","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Skalse, N","venue":"cs.LG","work_id":"b9869329-2719-4e51-889d-b637ee5e468d","year":2022},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2209.13085","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:cccf77f656af8c74df61d7aff544cd71ecc9e7911ae0ff0b05c8d1ad33c223d0","observation_id":"bfb8015f-5e33-47bc-999f-1209b38f7bdf","resolution":{"observed_at":"2026-05-10T19:00:45.572142Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Scaling laws for reward model overoptimization","venue":null,"work_id":"19dca190-8e4f-4d48-8365-3745a7f65cfe","year":2023},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:4f71d1b358c9bb13e48a6a1133fa50f498b86f612eaa2022c5d496b95ee66539","observation_id":"2101020d-97fa-4020-9d94-224294c23fdf","resolution":{"observed_at":"2026-05-16T13:22:55.863900Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2507.12399","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Roc-n-reroll: How verifier imperfection affects test-time scaling","venue":null,"work_id":"6647d287-2088-4c7a-b442-37a3621ea702","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:7c67a3c6cd88f1589986e4b71d73ecb1f8fc3da370f2a42e11f7688f621268c0","observation_id":"2e819090-0377-429c-8f40-bf42889e2576","resolution":{"observed_at":"2026-05-10T23:45:52.989245Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-07T10:53:40.936134Z","title":"TRL: Transformers Reinforcement Learning","venue":null,"work_id":"6e7d325d-c920-4461-9919-8108e384812e","year":2020},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:151164244f9db55ec536989395cfdffba872d65eacb27fc3b3f4cff0051e71dd","observation_id":"425e4519-bee4-4879-9351-fc61e21d47a4","resolution":{"observed_at":"2026-05-16T13:22:55.858818Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2512.13961","last_updated":"2026-04-14T15:12:44Z","snapshot_observed_at":"2026-08-16T16:18:35.731432Z","submitted_at":"2025-12-15T23:41:48Z","title":"Olmo 3","version":2},"cited_work":{"arxiv_id":"2512.13961","doi":"10.1007/s13755-026-00428-z","metadata_source":"pith","pith_arxiv_id":"2512.13961","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"Olmo 3","venue":"cs.CL","work_id":"74de5f5e-0a69-4f73-862d-e5705fa9f4bb","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2512.13961","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:ea8239c43b09e71c4bc3c1c5846625dba9605273bb46194a83a723257228ce5f","observation_id":"33d62132-cfd7-49b1-8b6b-a3cc2e0494b7","resolution":{"observed_at":"2026-05-10T23:45:53.027215Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"langdetect","venue":null,"work_id":"216ed148-aab5-43f6-8c62-040dbdf5a5d5","year":2021},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:94dd53981cc86ed63d58b073b65b7203b1d2ac3339e28a316e9a26a0cedb89b9","observation_id":"6fc3895d-9e6b-4ee6-a934-c175b8e1e015","resolution":{"observed_at":"2026-05-16T13:22:55.871163Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gonzalez, Hao Zhang, and Ion Stoica","venue":null,"work_id":"52a458b7-066f-4f6a-87b2-1bfa236cde69","year":2023},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:f6e9598ef38ea58079fb8ff1aee7fdf4889eac36d30add55efd788bf89427410","observation_id":"44a3a9e5-c032-45a3-99be-5a4096b36879","resolution":{"observed_at":"2026-05-16T13:22:55.873708Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-15T09:20:50.678596Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR"},"reference_resolution":{"displayed":34,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":0,"verified_exact":24,"verified_fuzzy":9},"total_outbound_references":34},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 34 of 34 outbound references and 1 inbound Pith citation observation for arXiv:2605.02909."}