{"as_of":"2026-08-20T23:04:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e4bc13327f3ff793b4cea41e3f4faaae2c9a0b2213177c830855c056a06a9483","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":61,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":61,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-20T06:33:59.587034+00:00","state":"measured"},{"denominator":61,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":61,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T11:53:28.847092Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2503.16419","last_updated":"2025-08-21T19:14:40Z","snapshot_observed_at":"2026-08-11T13:10:23.709172Z","submitted_at":"2025-03-20T17:59:38Z","title":"Stop Overthinking: A Survey on Efficient Reasoning for Large Language Models","version":4},"reference_index":191,"source":"pdf_text","source_observed_at":"2026-05-14T01:29:56.480020Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2503.16419"},"observation_digest":"sha256:fa39c32c181c16d60e1d580a60bed2b6c398edd060d3bf95573bf022dcbfd80b","observation_id":"3a06483f-e30c-4c86-a0d8-fc1f430aa4c1","resolution":{"observed_at":"2026-05-14T01:29:57.071592Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-16T11:53:28.847092Z","title":"arXiv preprint arXiv:2503.10460 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2504.14363","last_updated":"2025-07-05T09:43:33Z","snapshot_observed_at":"2026-08-20T04:00:34.570144Z","submitted_at":"2025-04-19T17:40:04Z","title":"Improving RL Exploration for LLM Reasoning through Retrospective Replay","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-16T11:53:28.847092Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2504.14363"},"observation_digest":"sha256:fbd1ff987ab84ccf9fb2df82872e8cc9a49e09064084dc5cbf82c59456be13d2","observation_id":"61e3575d-57e1-43c4-ac9a-d246f320caf0","resolution":{"observed_at":"2026-08-16T11:53:28.847092Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2504.14945","last_updated":"2025-06-22T00:18:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-21T08:09:13Z","title":"Learning to Reason under Off-Policy Guidance","version":5},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-15T23:17:02.701393Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2504.14945"},"observation_digest":"sha256:ea1242675aca743b419e087d8e43a4ec804dcdf94c1484677f516d1398fc9538","observation_id":"616e83c4-3352-43ef-b983-289c2dbd6c2a","resolution":{"observed_at":"2026-05-15T23:17:02.869340Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-16T10:39:31.300027Z","title":"Light-r1: Curriculum sft, dpo and rl for long cot from scratch and beyond","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2504.17565","last_updated":"2025-05-13T07:43:57Z","snapshot_observed_at":"2026-08-20T22:32:33.373295Z","submitted_at":"2025-04-24T13:57:53Z","title":"DeepDistill: Enhancing LLM Reasoning Capabilities via Large-Scale Difficulty-Graded Data Training","version":3},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-16T10:39:31.300027Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2504.17565"},"observation_digest":"sha256:1d14fb876bb7759497b6d5b422f1b2012e9386ab40798964a08e7ccfffc84d08","observation_id":"03557996-dbda-41b0-8670-3f7c8f9d161f","resolution":{"observed_at":"2026-08-16T10:39:31.300027Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2504.20571","last_updated":"2025-10-24T10:02:36Z","snapshot_observed_at":"2026-08-15T17:20:54.840134Z","submitted_at":"2025-04-29T09:24:30Z","title":"Reinforcement Learning for Reasoning in Large Language Models with One Training Example","version":3},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-15T19:51:04.779597Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2504.20571"},"observation_digest":"sha256:167d167a86c8be709adec54cad79ba54fd69ac887352b35ed8f6afeab82da6ed","observation_id":"89e88595-6b82-4a51-ba4e-fabbdd4e0f50","resolution":{"observed_at":"2026-05-15T19:51:05.002577Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-16T04:43:44.576915Z","title":"Light-r1: Curriculum sft, dpo and rl for long cot from scratch and beyond","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.00551","last_updated":"2025-05-15T14:16:03Z","snapshot_observed_at":"2026-08-17T18:29:23.516617Z","submitted_at":"2025-05-01T14:28:35Z","title":"100 Days After DeepSeek-R1: A Survey on Replication Studies and More Directions for Reasoning Language Models","version":3},"reference_index":131,"source":"arxiv_source","source_observed_at":"2026-08-16T04:43:44.576915Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2505.00551"},"observation_digest":"sha256:7c74ff96f4de6521d6fbeb084ace47a8110319ed199b11d9a7a03d6ae8595f3b","observation_id":"9c66fe2a-5ebb-41b7-b8f2-e87660e5ff04","resolution":{"observed_at":"2026-08-16T04:43:44.576915Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-15T23:56:46.331840Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.03469","last_updated":"2025-05-21T06:17:56Z","snapshot_observed_at":"2026-08-20T14:00:28.683849Z","submitted_at":"2025-05-06T12:18:11Z","title":"Long-Short Chain-of-Thought Mixture Supervised Fine-Tuning Eliciting Efficient Reasoning in Large Language Models","version":2},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-15T23:56:46.331840Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2505.03469"},"observation_digest":"sha256:433cfaa4905138699d668feb6b78d94845e8a03186f8c6f90304f8ccdc89207f","observation_id":"7ce99a49-f25f-48f4-95fe-c7949ed87f8b","resolution":{"observed_at":"2026-08-15T23:56:46.331840Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-15T21:07:08.171603Z","title":"Light-r1: Curriculum sft, DPO and RL for long COT from scratch and beyond.CoRR, abs/2503.10460, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.10937","last_updated":"2025-05-16T07:15:30Z","snapshot_observed_at":"2026-08-20T07:33:33.868637Z","submitted_at":"2025-05-16T07:15:30Z","title":"Reasoning with OmniThought: A Large CoT Dataset with Verbosity and Cognitive Difficulty Annotations","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-15T21:07:08.171603Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2505.10937"},"observation_digest":"sha256:8ab077af3db704cbb67609f990f24000e450fe952a3423b42c298be33d9c64d3","observation_id":"5d6df71d-8774-429f-9625-23df6d8e680a","resolution":{"observed_at":"2026-08-15T21:07:08.171603Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-15T20:48:44.462106Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.12043","last_updated":"2025-05-20T02:37:36Z","snapshot_observed_at":"2026-08-20T10:11:18.930400Z","submitted_at":"2025-05-17T15:12:47Z","title":"MoL for LLMs: Dual-Loss Optimization to Enhance Domain Expertise While Preserving General Capabilities","version":2},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-15T20:48:44.462106Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2505.12043"},"observation_digest":"sha256:7f2014846d8c70855e99d3edd61c4c25aaa8e30d29cecdc5c17c7540471019a3","observation_id":"4f4767c9-cb2f-4298-bdf5-5277d918943c","resolution":{"observed_at":"2026-08-15T20:48:44.462106Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-15T20:34:54.346546Z","title":"Light-r1: Curriculum sft, dpo and rl for long cot from scratch and beyond.arXiv preprint arXiv:2503.10460, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.12697","last_updated":"2025-05-19T04:37:53Z","snapshot_observed_at":"2026-08-16T06:45:16.946059Z","submitted_at":"2025-05-19T04:37:53Z","title":"Towards A Generalist Code Embedding Model Based On Massive Data Synthesis","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-15T20:34:54.346546Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2505.12697"},"observation_digest":"sha256:6bf6666b2b745f4cf2181aafab8ef9d8d6dbbb5f90b10aa80640e6cc7ce6754e","observation_id":"c883251e-66ba-41a3-ba3f-4962c134865a","resolution":{"observed_at":"2026-08-15T20:34:54.346546Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-07T15:06:35.469350Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond.arXiv preprint arXiv:2503.10460, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.16400","last_updated":"2025-06-05T17:59:12Z","snapshot_observed_at":"2026-08-17T12:06:20.372156Z","submitted_at":"2025-05-22T08:50:47Z","title":"AceReason-Nemotron: Advancing Math and Code Reasoning through Reinforcement Learning","version":3},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T15:06:35.469350Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2505.16400"},"observation_digest":"sha256:b16f64032ae4847c655108178311d295178fa9ea73c9eaea4ece9764712eff78","observation_id":"51031af8-96fb-4a71-ae2a-093c23cca5ee","resolution":{"observed_at":"2026-08-07T15:06:35.469350Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-07T15:05:18.720509Z","title":"Light-r1: Curriculum sft, dpo and rl for long cot from scratch and beyond","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17153","last_updated":"2025-05-22T11:27:01Z","snapshot_observed_at":"2026-08-14T21:35:27.351615Z","submitted_at":"2025-05-22T11:27:01Z","title":"Amplify Adjacent Token Differences: Enhancing Long Chain-of-Thought Reasoning with Shift-FFN","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T15:05:18.720509Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2505.17153"},"observation_digest":"sha256:5f13699c596f840810d41a8d77b2696629c9287a302855662ddfed88acad3eac","observation_id":"8a891a60-d82e-4554-820c-e3eec9d5f64f","resolution":{"observed_at":"2026-08-07T15:05:18.720509Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-07T14:47:58.110674Z","title":"Light-r1: Curriculum sft, dpo and rl for long cot from scratch and beyond","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17667","last_updated":"2025-05-27T09:39:47Z","snapshot_observed_at":"2026-08-18T11:46:25.087648Z","submitted_at":"2025-05-23T09:31:55Z","title":"QwenLong-L1: Towards Long-Context Large Reasoning Models with Reinforcement Learning","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:58.110674Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2505.17667"},"observation_digest":"sha256:c200b603660ececb361a79b2d0908eb639d91def3910c7b1dd91b5a4a11414e2","observation_id":"9b3a869c-772d-4d03-b820-8926f46311ae","resolution":{"observed_at":"2026-08-07T14:47:58.110674Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-07T14:38:46.973811Z","title":"Light-r1: Curriculum sft, dpo and rl for long cot from scratch and beyond, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.18086","last_updated":"2025-05-23T16:43:03Z","snapshot_observed_at":"2026-08-17T02:33:09.584278Z","submitted_at":"2025-05-23T16:43:03Z","title":"Stable Reinforcement Learning for Efficient Reasoning","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T14:38:46.973811Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2505.18086"},"observation_digest":"sha256:d210d0b25dd420f6f7ec894636d14c8ad0302f72d6d52e95111cc68ccc87eb9c","observation_id":"8f6d38db-9d42-4123-b1f9-235377011f6b","resolution":{"observed_at":"2026-08-07T14:38:46.973811Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-07T14:31:12.891931Z","title":"Light-r1: Curriculum sft, dpo and rl for long cot from scratch and beyond","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.18536","last_updated":"2025-05-24T06:01:48Z","snapshot_observed_at":"2026-08-16T23:13:31.947741Z","submitted_at":"2025-05-24T06:01:48Z","title":"Reinforcement Fine-Tuning Powers Reasoning Capability of Multimodal Large Language Models","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T14:31:12.891931Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2505.18536"},"observation_digest":"sha256:a1c9bd8ce3af029812c8925c51b436b484d8dc94ea9d189ca3de46dee4f78c3e","observation_id":"58bcbb80-eed5-449a-9c08-cdb14c4ad7e0","resolution":{"observed_at":"2026-08-07T14:31:12.891931Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-07T14:23:09.386204Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.19126","last_updated":"2025-05-25T12:47:39Z","snapshot_observed_at":"2026-08-15T17:36:19.131323Z","submitted_at":"2025-05-25T12:47:39Z","title":"MMATH: A Multilingual Benchmark for Mathematical Reasoning","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-07T14:23:09.386204Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2505.19126"},"observation_digest":"sha256:9d4877ef60d73bf567b34bf7ed8e4af8296a7f0d9ffa010a05f8d09fca01fa05","observation_id":"d83f6548-a7f5-4fd5-aba7-4754a97c478f","resolution":{"observed_at":"2026-08-07T14:23:09.386204Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-07T14:12:19.947222Z","title":"Light-r1: Curriculum sft, dpo and rl for long cot from scratch and beyond,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.19716","last_updated":"2025-05-26T09:04:44Z","snapshot_observed_at":"2026-08-17T15:44:31.687856Z","submitted_at":"2025-05-26T09:04:44Z","title":"Concise Reasoning, Big Gains: Pruning Long Reasoning Trace with Difficulty-Aware Prompting","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T14:12:19.947222Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2505.19716"},"observation_digest":"sha256:cf352b8c8d3df50fc98ac3d1fa19714b27686de2e518bd68c02d7b7d5430caad","observation_id":"857e40f0-629c-4ec6-a05c-83cfacac4271","resolution":{"observed_at":"2026-08-07T14:12:19.947222Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-07T14:13:11.080617Z","title":"Light-r1: Curriculum sft, DPO and RL for long COT from scratch and beyond.CoRR, abs/2503.10460, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.19815","last_updated":"2025-05-26T10:52:17Z","snapshot_observed_at":"2026-08-14T19:11:50.543978Z","submitted_at":"2025-05-26T10:52:17Z","title":"Deciphering Trajectory-Aided LLM Reasoning: An Optimization Perspective","version":1},"reference_index":112,"source":"pdf_text","source_observed_at":"2026-08-07T14:13:11.080617Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2505.19815"},"observation_digest":"sha256:b1861bdf85ff7167fe155b17bb3c05535a83179552894527388df8c00cdc4541","observation_id":"d1d4155d-0ead-46e9-a738-5be0cf0490c0","resolution":{"observed_at":"2026-08-07T14:13:11.080617Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-07T14:11:08.796570Z","title":"Light-r1: Curriculum sft, dpo and rl for long cot from scratch and beyond.arXiv preprint arXiv:2503.10460, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.19914","last_updated":"2025-06-09T07:49:32Z","snapshot_observed_at":"2026-08-07T16:09:17.796555Z","submitted_at":"2025-05-26T12:40:31Z","title":"Enigmata: Scaling Logical Reasoning in Large Language Models with Synthetic Verifiable Puzzles","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T14:11:08.796570Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2505.19914"},"observation_digest":"sha256:4bd567accf3ed46289ffc6a52bbe25eb682c7b0ab72cf74e21196e6cfb1a38a8","observation_id":"28c8782f-2b2e-43bf-b38f-0383ac676e04","resolution":{"observed_at":"2026-08-07T14:11:08.796570Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-07T14:07:49.365734Z","title":"Light-r1: Curriculum sft, dpo and rl for long cot from scratch and beyond.arXiv preprint arXiv:2503.10460, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.19949","last_updated":"2025-05-26T13:15:26Z","snapshot_observed_at":"2026-08-20T03:49:01.362977Z","submitted_at":"2025-05-26T13:15:26Z","title":"Which Data Attributes Stimulate Math and Code Reasoning? An Investigation via Influence Functions","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T14:07:49.365734Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2505.19949"},"observation_digest":"sha256:bec8813fafef14c7b99bb74e51efa977b52d029f4d2794e863f1f267f8534118","observation_id":"6eb165d9-c577-4497-85c8-12b25fff21a5","resolution":{"observed_at":"2026-08-07T14:07:49.365734Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-07T13:41:30.443487Z","title":"Light-r1: Curriculum sft, dpo and rl for long cot from scratch and beyond","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.21178","last_updated":"2025-05-27T13:29:51Z","snapshot_observed_at":"2026-08-18T19:24:53.799085Z","submitted_at":"2025-05-27T13:29:51Z","title":"Walk Before You Run! Concise LLM Reasoning via Reinforcement Learning","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T13:41:30.443487Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2505.21178"},"observation_digest":"sha256:09dd133e1eab1fc142ebe0661344418f184acc1b642cd99c2354f71d0ab8abd7","observation_id":"a85ba105-22b8-4869-ad5a-b8688d9142cc","resolution":{"observed_at":"2026-08-07T13:41:30.443487Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2505.22312","last_updated":"2025-05-29T09:07:33Z","snapshot_observed_at":"2026-08-15T23:24:42.542733Z","submitted_at":"2025-05-28T12:56:04Z","title":"Skywork Open Reasoner 1 Technical Report","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-17T04:26:47.283983Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2505.22312"},"observation_digest":"sha256:be4f0f54c097b7cf46ab5229ad97dec389f6599b9f2ac528e94a82b0f10efb1c","observation_id":"d906625e-03ee-4bc7-90cf-733794c7e472","resolution":{"observed_at":"2026-05-17T04:26:47.350497Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-07T13:05:27.038237Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.22756","last_updated":"2025-05-28T18:18:49Z","snapshot_observed_at":"2026-08-17T01:33:15.660380Z","submitted_at":"2025-05-28T18:18:49Z","title":"Decomposing Elements of Problem Solving: What \"Math\" Does RL Teach?","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:27.038237Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2505.22756"},"observation_digest":"sha256:958de69450bde13dbeaae056065df23bdf2b942f3acd817617905fb2ab4913d2","observation_id":"6d56d6f1-1820-4fc9-865c-f168dcb42c1c","resolution":{"observed_at":"2026-08-07T13:05:27.038237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-07T12:59:21.956041Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.23091","last_updated":"2025-06-23T08:47:25Z","snapshot_observed_at":"2026-08-16T08:04:58.307699Z","submitted_at":"2025-05-29T04:51:56Z","title":"Infi-MMR: Curriculum-based Unlocking Multimodal Reasoning via Phased Reinforcement Learning in Multimodal Small Language Models","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T12:59:21.956041Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2505.23091"},"observation_digest":"sha256:4435c86fc480204e736f35d4af75846e8b87065456f3b8a7a940b949782d297e","observation_id":"40d1d414-ca7a-4e9a-8d5a-2ee63473299f","resolution":{"observed_at":"2026-08-07T12:59:21.956041Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-07T11:17:37.235199Z","title":"Light-R1: Curriculum sft, dpo and rl for long cot from scratch and beyond.arXiv preprint arXiv:2503.10460, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.03077","last_updated":"2025-06-03T16:54:15Z","snapshot_observed_at":"2026-08-16T17:48:46.961655Z","submitted_at":"2025-06-03T16:54:15Z","title":"StreamBP: Memory-Efficient Exact Backpropagation for Long Sequence Training of LLMs","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T11:17:37.235199Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2506.03077"},"observation_digest":"sha256:2e8c7ec04c912b89dd4fa591d3f89e78f25f690c1b917021b1b9e885caabc02b","observation_id":"ec882aa4-946f-4bb0-924e-0ca1319546e9","resolution":{"observed_at":"2026-08-07T11:17:37.235199Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-07T10:54:07.838570Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.04065","last_updated":"2025-06-04T15:31:46Z","snapshot_observed_at":"2026-08-13T19:00:44.799706Z","submitted_at":"2025-06-04T15:31:46Z","title":"Progressive Mastery: Customized Curriculum Learning with Guided Prompting for Mathematical Reasoning","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-07T10:54:07.838570Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2506.04065"},"observation_digest":"sha256:7d4d0cb8162bbe13ebfe6a379e48f6c2e8c1cf7885290d4be1933bde7bb65ef8","observation_id":"235ad0e0-a4d8-4234-93ea-98d1f2f340ca","resolution":{"observed_at":"2026-08-07T10:54:07.838570Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-07T05:01:24.233618Z","title":"Light-r1: Curriculum sft, dpo and rl for long cot from scratch and beyond.arXiv preprint arXiv:2503.10460, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.08989","last_updated":"2025-06-10T17:02:00Z","snapshot_observed_at":"2026-08-19T22:56:54.842002Z","submitted_at":"2025-06-10T17:02:00Z","title":"SwS: Self-aware Weakness-driven Problem Synthesis in Reinforcement Learning for LLM Reasoning","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T05:01:24.233618Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2506.08989"},"observation_digest":"sha256:2a27872b8471f473b6badc344676c54d3f94a3f26a8ed1284984af22c6460b25","observation_id":"bf457c15-fa3a-4d33-a7ad-bcdc482e83be","resolution":{"observed_at":"2026-08-07T05:01:24.233618Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-07T01:05:26.848970Z","title":"Light-r1: Curriculum sft, dpo and rl for long cot from scratch and beyond","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.11986","last_updated":"2025-06-13T17:46:02Z","snapshot_observed_at":"2026-08-20T09:16:57.916280Z","submitted_at":"2025-06-13T17:46:02Z","title":"Schema-R1: A reasoning training approach for schema linking in Text-to-SQL Task","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T01:05:26.848970Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2506.11986"},"observation_digest":"sha256:2b2d1e9f5a681ff12adf6e6abff79894450226dc50d2ece81eddb565f2cb7433","observation_id":"39536fc5-73e8-4b6e-b543-381f67e043d0","resolution":{"observed_at":"2026-08-07T01:05:26.848970Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-07T00:41:36.113617Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond.arXiv preprint arXiv:2503.10460, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.13284","last_updated":"2025-06-16T09:27:48Z","snapshot_observed_at":"2026-08-13T16:04:27.350434Z","submitted_at":"2025-06-16T09:27:48Z","title":"AceReason-Nemotron 1.1: Advancing Math and Code Reasoning through SFT and RL Synergy","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T00:41:36.113617Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2506.13284"},"observation_digest":"sha256:749e8434ad64409b243ddf3e938c5efaf921734238d5b9797f93971ace446c69","observation_id":"71a6e346-aac4-4dea-a418-2222643eb1ae","resolution":{"observed_at":"2026-08-07T00:41:36.113617Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-06T22:59:47.990084Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.20241","last_updated":"2025-06-25T08:36:12Z","snapshot_observed_at":"2026-08-13T16:56:05.130404Z","submitted_at":"2025-06-25T08:36:12Z","title":"Enhancing Large Language Models through Structured Reasoning","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T22:59:47.990084Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2506.20241"},"observation_digest":"sha256:8a4706b1e13723d73513b1dade25917c954a4848f86aeed48e870bf9e4842349","observation_id":"2c88e462-fd60-4c69-a1e3-02b0b28a2579","resolution":{"observed_at":"2026-08-06T22:59:47.990084Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-15T18:39:18.971569Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.00045","last_updated":"2025-06-23T22:05:21Z","snapshot_observed_at":"2026-08-20T17:33:27.320921Z","submitted_at":"2025-06-23T22:05:21Z","title":"CaughtCheating: Is Your MLLM a Good Cheating Detective? Exploring the Boundary of Visual Perception and Reasoning","version":1},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-08-15T18:39:18.971569Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2507.00045"},"observation_digest":"sha256:ecda2dea6424b59acd3b5ca616393e7cb430c2687486db598c8de75bcf75b814","observation_id":"8cb3ee36-7012-4e15-b16c-47b1fc5031b4","resolution":{"observed_at":"2026-08-15T18:39:18.971569Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-06T20:46:22.631170Z","title":"Light-r1: Curriculum sft, dpo and rl for long cot from scratch and beyond.arXiv preprint arXiv:2503.10460, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.01887","last_updated":"2025-07-02T16:57:01Z","snapshot_observed_at":"2026-08-14T13:08:06.376797Z","submitted_at":"2025-07-02T16:57:01Z","title":"MiCoTA: Bridging the Learnability Gap with Intermediate CoT and Teacher Assistants","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T20:46:22.631170Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2507.01887"},"observation_digest":"sha256:6fac475c0742345e60f6261f8459b3456e471fd1be3c0b89ff0353fa5d8677c1","observation_id":"c17a0cd6-9280-4801-9636-f0b99cb58c8d","resolution":{"observed_at":"2026-08-06T20:46:22.631170Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-06T18:57:38.720506Z","title":"Light-r1: Curriculum sft, DPO and RL for long COT from scratch and beyond","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.06892","last_updated":"2025-07-11T10:32:34Z","snapshot_observed_at":"2026-08-17T17:33:02.005399Z","submitted_at":"2025-07-09T14:29:45Z","title":"Squeeze the Soaked Sponge: Efficient Off-policy Reinforcement Finetuning for Large Language Model","version":3},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-06T18:57:38.720506Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2507.06892"},"observation_digest":"sha256:a53233d6f052dbc837b06d8d16d112deb06fa9ca4fa92c8fbdb072b037dce372","observation_id":"f4d0928f-6f77-40d6-9eaf-e34070df6e62","resolution":{"observed_at":"2026-08-06T18:57:38.720506Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-06T18:26:08.525387Z","title":"Light-r1: Curriculum sft, dpo and rl for long cot from scratch and beyond, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.08267","last_updated":"2025-07-11T02:26:01Z","snapshot_observed_at":"2026-08-18T08:05:53.035856Z","submitted_at":"2025-07-11T02:26:01Z","title":"A Practical Two-Stage Recipe for Mathematical LLMs: Maximizing Accuracy with SFT and Efficiency with Reinforcement Learning","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-06T18:26:08.525387Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2507.08267"},"observation_digest":"sha256:4430e4ceba5edb866707ea793f06bf49f647a0478a4222bcb23a7c3b91c7b85b","observation_id":"6c52c4c3-7204-4084-8ba8-a61935fa4f62","resolution":{"observed_at":"2026-08-06T18:26:08.525387Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-06T17:54:17.847748Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.09662","last_updated":"2025-07-13T14:51:59Z","snapshot_observed_at":"2026-08-20T12:31:43.651320Z","submitted_at":"2025-07-13T14:51:59Z","title":"Towards Concise and Adaptive Thinking in Large Reasoning Models: A Survey","version":1},"reference_index":200,"source":"arxiv_source","source_observed_at":"2026-08-06T17:54:17.847748Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2507.09662"},"observation_digest":"sha256:8995e6ae53d336e1814608ef90c22ac9ac9ac2e37e75ef189559244811a9d0b0","observation_id":"d4364aea-4370-4dbf-bd1e-802911a48418","resolution":{"observed_at":"2026-08-06T17:54:17.847748Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-06T17:35:31.692436Z","title":"Light-r1: Curriculum sft, dpo and rl for long cot from scratch and beyond.arXiv preprint arXiv:2503.10460, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.10541","last_updated":"2025-07-15T06:16:53Z","snapshot_observed_at":"2026-08-19T20:11:31.924910Z","submitted_at":"2025-07-14T17:58:47Z","title":"REST: Stress Testing Large Reasoning Models by Asking Multiple Problems at Once","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T17:35:31.692436Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2507.10541"},"observation_digest":"sha256:6f3b3c7ae7ed636eba03e29d490fdf28a06f6dc46723a3695babe5da30fbac20","observation_id":"b60c03f1-14c4-4a95-a520-91ce8f4fccf3","resolution":{"observed_at":"2026-08-06T17:35:31.692436Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-06T14:53:04.599526Z","title":"Light-r1: Curriculum sft, dpo and rl for long cot from scratch and beyond","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.17512","last_updated":"2025-07-23T13:51:04Z","snapshot_observed_at":"2026-08-09T23:40:29.433863Z","submitted_at":"2025-07-23T13:51:04Z","title":"Can One Domain Help Others? A Data-Centric Study on Multi-Domain Reasoning via Reinforcement Learning","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T14:53:04.599526Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2507.17512"},"observation_digest":"sha256:d01c8676411db8a9b9b2421dd315ef807f2e3362a67f7dbc8d53bcdc416df37d","observation_id":"f940d917-eda9-44d6-8168-e3c54efb1326","resolution":{"observed_at":"2026-08-06T14:53:04.599526Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2508.08636","last_updated":"2026-05-20T08:06:05Z","snapshot_observed_at":"2026-08-16T10:26:53.150702Z","submitted_at":"2025-08-12T05:00:00Z","title":"InternBootcamp Technical Report: Boosting LLM Reasoning with Verifiable Task Scaling","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-21T22:33:09.674822Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2508.08636"},"observation_digest":"sha256:f1f781f78976b52dfd0d60ef8a462032bf49d7dcc57923e82e12f5f0a4c1314a","observation_id":"ba84593e-9ae9-4427-8d11-3e07adbfd0b4","resolution":{"observed_at":"2026-05-21T22:34:23.988237Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T16:19:53.921525Z","title":"Light-r1: Curriculum sft, dpo and rl for long cot from scratch and beyond.arXiv preprint arXiv:2503.10460, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.18773","last_updated":"2025-08-26T07:57:28Z","snapshot_observed_at":"2026-08-16T14:45:23.839099Z","submitted_at":"2025-08-26T07:57:28Z","title":"ThinkDial: An Open Recipe for Controlling Reasoning Effort in Large Language Models","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-05T16:19:53.921525Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2508.18773"},"observation_digest":"sha256:03921f4a8c8a35fe2ae9ca632920f94a466ab8b8415ed14907c0792e56cd58e8","observation_id":"c5b6fc67-3e29-4c6a-bd9e-45e90dcbf817","resolution":{"observed_at":"2026-08-05T16:19:53.921525Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T15:11:01.110314Z","title":"arXiv:2503.10460","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.20384","last_updated":"2025-08-28T03:16:15Z","snapshot_observed_at":"2026-08-14T19:11:49.873888Z","submitted_at":"2025-08-28T03:16:15Z","title":"Uncertainty Under the Curve: A Sequence-Level Entropy Area Metric for Reasoning LLM","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-05T15:11:01.110314Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2508.20384"},"observation_digest":"sha256:e44bf30fcbfaa0d22fdbf5ad0547fa5409493a4d92f0d5d865a641ca0f49f4af","observation_id":"d205813b-567a-43ec-9793-c23f36ab7807","resolution":{"observed_at":"2026-08-05T15:11:01.110314Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T05:45:04.041310Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.05007","last_updated":"2025-09-08T03:26:03Z","snapshot_observed_at":"2026-08-16T11:58:24.387495Z","submitted_at":"2025-09-05T11:14:11Z","title":"Sticker-TTS: Learn to Utilize Historical Experience with a Sticker-driven Test-Time Scaling Framework","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-05T05:45:04.041310Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2509.05007"},"observation_digest":"sha256:2e797c24b0f339d81b0d2da23498b6c7f712c5e67cb456c92f7fe4d7968b185a","observation_id":"08fe2101-9ae3-4671-9a47-74cd828efa0a","resolution":{"observed_at":"2026-08-05T05:45:04.041310Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T05:45:04.085149Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.05007","last_updated":"2025-09-08T03:26:03Z","snapshot_observed_at":"2026-08-16T11:58:24.387495Z","submitted_at":"2025-09-05T11:14:11Z","title":"Sticker-TTS: Learn to Utilize Historical Experience with a Sticker-driven Test-Time Scaling Framework","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-05T05:45:04.085149Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2509.05007"},"observation_digest":"sha256:1634990f7f0a39f53ca2703ff1abd21f00919af5fac52f848580741065d3c9bb","observation_id":"2df28670-0e00-44f9-a3c3-8772a2db9d8a","resolution":{"observed_at":"2026-08-05T05:45:04.085149Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-04T23:23:41.779153Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.06650","last_updated":"2025-09-08T13:04:07Z","snapshot_observed_at":"2026-08-15T01:46:49.069429Z","submitted_at":"2025-09-08T13:04:07Z","title":"Domain-Aware RAG: MoL-Enhanced RL for Efficient Training and Scalable Retrieval","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-04T23:23:41.779153Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2509.06650"},"observation_digest":"sha256:c60cbdb5843552a3580d319564d320a9ca9eff4d70e90748dd0078e6de37a977","observation_id":"e1c883b7-520d-4b4a-97f1-d8b690a827bc","resolution":{"observed_at":"2026-08-04T23:23:41.779153Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2509.25758","last_updated":"2026-04-14T15:45:09Z","snapshot_observed_at":"2026-08-15T15:08:48.009649Z","submitted_at":"2025-09-30T04:23:43Z","title":"Thinking Sparks!: Emergent Attention Heads in Reasoning Models During Post Training","version":2},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-05-18T13:28:32.093512Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2509.25758"},"observation_digest":"sha256:327b2608531d11f223b51990a83f9eebe1fc467532d1bd710a7b26a743c89284","observation_id":"400fda6c-c35b-4b51-92e0-b5081c768edb","resolution":{"observed_at":"2026-05-18T13:31:24.978441Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2510.03988","last_updated":"2026-04-14T20:09:39Z","snapshot_observed_at":"2026-08-11T23:53:09.972171Z","submitted_at":"2025-10-05T01:15:32Z","title":"The Signal is in the Steps: Local Scoring for Reasoning Data Selection","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-18T10:14:27.739531Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2510.03988"},"observation_digest":"sha256:d0995cc07101c277edb1151fdb718fb1430ab75fcd8f8b36f9364db28497d458","observation_id":"b7b7f24b-3eea-4265-9b27-09e63c6e8bb5","resolution":{"observed_at":"2026-05-18T10:16:14.172268Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2510.04265","last_updated":"2026-05-12T01:55:07Z","snapshot_observed_at":"2026-08-13T07:36:21.999871Z","submitted_at":"2025-10-05T16:14:03Z","title":"Don't Pass@k: A Bayesian Framework for Large Language Model Evaluation","version":4},"reference_index":99,"source":"pdf_text","source_observed_at":"2026-05-18T10:04:39.223895Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2510.04265"},"observation_digest":"sha256:0e9609da907331c5d77f84c78ead7ee04aa2e202e8c0d9b52a97775dc5728a0f","observation_id":"6eb96528-8a20-4626-b3e4-0e81d6459750","resolution":{"observed_at":"2026-05-18T10:06:13.802986Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2510.18814","last_updated":"2026-07-05T03:44:20Z","snapshot_observed_at":"2026-08-19T21:15:13.597413Z","submitted_at":"2025-10-21T17:15:56Z","title":"A Model Can Help Itself: Reward-Free Self-Training for LLM Reasoning","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-18T05:11:29.205366Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2510.18814"},"observation_digest":"sha256:3f56d528dcf448527f0f6c2d120c044a859494ea0407278869e635541cad0bdf","observation_id":"15289bb2-e131-49f9-ae78-6e8beb41fd51","resolution":{"observed_at":"2026-05-18T05:12:23.662328Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2510.18814","last_updated":"2026-07-05T03:44:20Z","snapshot_observed_at":"2026-08-19T21:15:13.597413Z","submitted_at":"2025-10-21T17:15:56Z","title":"A Model Can Help Itself: Reward-Free Self-Training for LLM Reasoning","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-21T20:06:16.172916Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2510.18814"},"observation_digest":"sha256:861512d8b1ed2189449d48a357dec85d6af70845c86885b4ee2c0e94089d78a4","observation_id":"be7e2720-b16e-4fdb-ae74-588e4da07b09","resolution":{"observed_at":"2026-05-21T20:10:34.891829Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-04T08:53:07.747822Z","title":"Light-R1: Curriculum sft, dpo and rl for long cot from scratch and beyond.arXiv preprint arXiv:2503.10460,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2510.18814","last_updated":"2026-07-05T03:44:20Z","snapshot_observed_at":"2026-08-19T21:15:13.597413Z","submitted_at":"2025-10-21T17:15:56Z","title":"A Model Can Help Itself: Reward-Free Self-Training for LLM Reasoning","version":4},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-04T08:53:07.747822Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2510.18814"},"observation_digest":"sha256:62516d5c40444648c521764b590f9b5fd4c91774c0d7c608c0da35ca03b2db02","observation_id":"a4abdab7-eff5-47e4-8935-3ee46b435881","resolution":{"observed_at":"2026-08-04T08:53:07.747822Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2512.03847","last_updated":"2026-05-06T14:15:19Z","snapshot_observed_at":"2026-08-16T02:48:57.063506Z","submitted_at":"2025-12-03T14:48:38Z","title":"DVPO: Distributional Value Modeling-based Policy Optimization for LLM Post-Training","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-17T01:46:21.744857Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2512.03847"},"observation_digest":"sha256:3b3035f6fbfe5ce266749083146594fda4cd6747ba22ec6eb752f333573bc0ab","observation_id":"d14952ea-21c3-4e8c-8ef4-34e0280d2194","resolution":{"observed_at":"2026-05-17T01:48:50.926478Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2602.01970","last_updated":"2026-05-15T12:23:06Z","snapshot_observed_at":"2026-08-18T19:27:56.462164Z","submitted_at":"2026-02-02T11:24:36Z","title":"Small Generalizable Prompt Predictive Models Can Steer Efficient RL Post-Training of Large Reasoning Models","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-21T14:09:26.842696Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2602.01970"},"observation_digest":"sha256:d304fe135ae775d942bbf63fa6603b4f2f16bdb29b4fbc15c7d9a500645f6f55","observation_id":"1bf0c8de-d4d5-449c-998f-c000e8621a0d","resolution":{"observed_at":"2026-05-21T14:10:13.173355Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2604.17614","last_updated":"2026-04-19T20:58:25Z","snapshot_observed_at":"2026-08-14T08:41:23.060090Z","submitted_at":"2026-04-19T20:58:25Z","title":"Characterizing Model-Native Skills","version":1},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-05-10T05:42:49.694715Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2604.17614"},"observation_digest":"sha256:4e83626bc8237bc5ad766719ebb04ee03ad57ee6d87159e8c1c808992153ace7","observation_id":"09a06dac-043b-4ecc-bfb4-0a1933af2e20","resolution":{"observed_at":"2026-05-10T06:06:19.521433Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2605.06165","last_updated":"2026-05-07T12:51:49Z","snapshot_observed_at":"2026-08-17T00:27:13.360022Z","submitted_at":"2026-05-07T12:51:49Z","title":"Post Reasoning: Improving the Performance of Non-Thinking Models at No Cost","version":1},"reference_index":159,"source":"arxiv_source","source_observed_at":"2026-05-08T10:19:08.451445Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2605.06165"},"observation_digest":"sha256:57ba0d93ad5a3de2b57525aaf6384461c0c8704c381e6645287352b27878ae05","observation_id":"0e9be8f2-6c0b-4de7-98bb-6803cffdb2a4","resolution":{"observed_at":"2026-05-11T20:06:09.755531Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2605.08441","last_updated":"2026-05-08T20:03:19Z","snapshot_observed_at":"2026-08-11T08:24:15.849440Z","submitted_at":"2026-05-08T20:03:19Z","title":"DUET: Optimize Token-Budget Allocation for Reinforcement Learning with Verifiable Rewards","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-12T01:57:11.065744Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2605.08441"},"observation_digest":"sha256:44aa47c192baacd216e9e66a7f3ef44a7bd6b8523f8b79eaa634cde4efcf8e09","observation_id":"5a49ae54-1398-4a86-9a44-d95a03d6c9a1","resolution":{"observed_at":"2026-05-12T07:46:28.509523Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2605.08905","last_updated":"2026-05-09T11:57:25Z","snapshot_observed_at":"2026-08-11T11:04:38.944000Z","submitted_at":"2026-05-09T11:57:25Z","title":"Forge: Quality-Aware Reinforcement Learning for NP-Hard Optimization in LLMs","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-12T02:44:33.143247Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2605.08905"},"observation_digest":"sha256:01a3799eed062fb79f4445f9d272191abe49d34f3953fa5a1eac71ad18d1489e","observation_id":"e229cccd-8685-440f-a136-6c54e5289bd3","resolution":{"observed_at":"2026-05-12T02:46:18.839590Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2605.14054","last_updated":"2026-06-02T19:43:46Z","snapshot_observed_at":"2026-08-16T18:47:45.576255Z","submitted_at":"2026-05-13T19:23:53Z","title":"Bad Seeing or Bad Thinking? Rewarding Perception for Multimodal Reasoning","version":1},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-05-15T05:14:28.256032Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2605.14054"},"observation_digest":"sha256:e8d5d90b2e5b48e08ddfb728291b0479d96f818374908c055a11f72cbee923fc","observation_id":"1d6efd00-eb3d-4326-8317-629164fefbbe","resolution":{"observed_at":"2026-05-15T05:15:02.865013Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2605.22567","last_updated":"2026-05-21T14:47:52Z","snapshot_observed_at":"2026-08-16T17:36:50.516706Z","submitted_at":"2026-05-21T14:47:52Z","title":"LANG: Reinforcement Learning for Multilingual Reasoning with Language-Adaptive Hint Guidance","version":1},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-05-22T06:19:44.377733Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2605.22567"},"observation_digest":"sha256:b6d62d89c9aea65da35c802822bdf1aa56286b5626d73d57635a1e209e39df5b","observation_id":"d4a5b6d9-9cbb-40e4-9b29-12e08523d53f","resolution":{"observed_at":"2026-05-22T06:21:09.548667Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2606.01168","last_updated":"2026-05-31T11:20:00Z","snapshot_observed_at":"2026-08-15T12:13:34.288229Z","submitted_at":"2026-05-31T11:20:00Z","title":"Thinking Economically: A Hierarchical Framework for Adaptive-Complexity Reasoning in LLMs","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-06-28T17:05:48.244094Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2606.01168"},"observation_digest":"sha256:598a1d6adf36e8f33f02e209fea3109bc9b37aa61d888bccb3b9f12cee63cca8","observation_id":"ae67e4f2-82f4-4c79-bc4f-db0a819cbd93","resolution":{"observed_at":"2026-06-28T17:12:25.200187Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2606.11119","last_updated":"2026-06-09T17:16:03Z","snapshot_observed_at":"2026-08-18T17:54:27.819254Z","submitted_at":"2026-06-09T17:16:03Z","title":"TRACE: A Unified Rollout Budget Allocation Framework for Efficient Agentic Reinforcement Learning","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-06-27T13:55:35.363377Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2606.11119"},"observation_digest":"sha256:aae554c33c0d9f0914bc10abd25458724a6830b6e6360973c8d85d3bc56542ee","observation_id":"221004fe-29e5-487f-8700-2e681c0208f9","resolution":{"observed_at":"2026-07-03T04:27:36.717007Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2607.02234","last_updated":"2026-07-02T14:33:07Z","snapshot_observed_at":"2026-08-12T11:57:55.007718Z","submitted_at":"2026-07-02T14:33:07Z","title":"Purified OPSD: On-Policy Self-Distillation Without Losing How to Think","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-07-03T13:56:13.827493Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2607.02234"},"observation_digest":"sha256:d95d9d81127f7cf12fe3a93a0a4f39bdaa1c78110cb5a0017d7ddf88775b7556","observation_id":"ec1f147d-143b-4586-bbec-816290c2e37b","resolution":{"observed_at":"2026-07-03T13:58:21.051491Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-15T14:49:03.983309Z","title":"arXiv preprint arXiv:2503.10460 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04001","last_updated":"2026-08-04T17:57:20Z","snapshot_observed_at":"2026-08-17T15:56:44.886598Z","submitted_at":"2026-08-04T17:57:20Z","title":"Test-Time Scaling in Reasoning LLMs: Inference Regimes, Evaluation, and Reproducibility","version":1},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-08-15T14:49:03.983309Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2608.04001"},"observation_digest":"sha256:6dabe9f01f7ab13df536c2fc9acb374291f091e560e987a276c336ce9d4d30f0","observation_id":"1eafb797-f179-47da-97b1-418540031a12","resolution":{"observed_at":"2026-08-15T14:49:03.983309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2503.10460/citation-record","integrity":"/paper/2503.10460/integrity","json":"/paper/2503.10460/citation-record.json","paper":"/paper/2503.10460"},"outbound":[],"paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","latest_version":4,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-20T04:23:28.056352Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"thesis":"As of 20 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 61 inbound Pith citation observations for arXiv:2503.10460."}