{"as_of":"2026-08-09T23:47:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:4c54410af9cdd125ded95e10bf74e0023ac9945aa34bcbee8a002ddce9d2b02c","coverage":[{"denominator":53,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":53,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T10:31:37.980968Z","state":"measured"},{"denominator":54,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":54,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-18T00:02:24.352947Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-18T00:02:25.420148Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"cited_work":{"arxiv_id":"2509.04027","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2509.04027","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Cot-space: A theoretical framework for internal slow-thinking via reinforcement learning","venue":null,"work_id":"b79abe84-c74a-4d4f-9749-ce5c2dca82f1","year":2025},"citing_paper":{"arxiv_id":"2509.08827","last_updated":"2025-10-09T17:08:52Z","snapshot_observed_at":"2026-08-06T15:38:05.011922Z","submitted_at":"2025-09-10T17:59:43Z","title":"A Survey of Reinforcement Learning for Large Reasoning Models","version":3},"reference_index":152,"source":"arxiv_source","source_observed_at":"2026-05-18T00:02:24.352947Z"},"links":{"cited_paper":"/paper/2509.04027","citing_paper":"/paper/2509.08827"},"observation_digest":"sha256:e4f5fc544b5dc454edd3f0ae4c629218ff882282efcc9dc9a077f5fc17fde851","observation_id":"f8bb9d9c-cda0-4a4f-8558-1a01fd930ed4","resolution":{"observed_at":"2026-06-05T02:16:19.319482Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2509.04027/citation-record","integrity":"/paper/2509.04027/integrity","json":"/paper/2509.04027/citation-record.json","paper":"/paper/2509.04027"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:31:37.734744Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.734744Z"},"links":{"citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:728dea0450076a33fbc9a7ae71c474447d2b70ecedee894ff8f86e9fbc1405e5","observation_id":"653d22f5-53a2-4c5d-bd2d-c921f972abc5","resolution":{"observed_at":"2026-08-05T10:31:37.734744Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14740","last_updated":"2024-02-26T18:26:25Z","snapshot_observed_at":"2026-08-09T14:30:33.899591Z","submitted_at":"2024-02-22T17:52:34Z","title":"Back to Basics: Revisiting REINFORCE Style Optimization for Learning from Human Feedback in LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.14740","snapshot_observed_at":"2026-08-05T10:31:37.745578Z","title":"Back to basics: Revisiting reinforce style optimization for learning from human feedback in llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.745578Z"},"links":{"cited_paper":"/paper/2402.14740","citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:871a37efb86777f0e1d550dbfb2b2ae6c2d02668805d5de27be4dca148a72b92","observation_id":"e0ece3d5-b6d2-4ecf-b2e1-086f710063dc","resolution":{"observed_at":"2026-08-05T10:31:37.745578Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:31:37.751394Z","title":"Language models are few-shot learners","venue":null,"work_id":null,"year":1901},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.751394Z"},"links":{"citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:f3c4b50c4ac2592da4e286ea48a2f22ff50ecca78b124c15cf9072992382f87c","observation_id":"3a99df52-5d06-4985-9c17-1f152740cc5b","resolution":{"observed_at":"2026-08-05T10:31:37.751394Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09567","last_updated":"2025-07-18T15:57:54Z","snapshot_observed_at":"2026-08-08T22:33:20.124926Z","submitted_at":"2025-03-12T17:35:03Z","title":"Towards Reasoning Era: A Survey of Long Chain-of-Thought for Reasoning Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09567","snapshot_observed_at":"2026-08-05T10:31:37.755360Z","title":"Towards reasoning era: A survey of long chain-of-thought for reasoning large language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.755360Z"},"links":{"cited_paper":"/paper/2503.09567","citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:20276cc67cf6e7486e8605c48a0de4adb4802f58c60fb47fd3644d4fb7102def","observation_id":"7eadecf7-832c-4bab-b6de-39a6d38e2ae3","resolution":{"observed_at":"2026-08-05T10:31:37.755360Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2211.12588","last_updated":"2023-10-23T01:27:38Z","snapshot_observed_at":"2026-08-02T13:06:11.850456Z","submitted_at":"2022-11-22T21:06:00Z","title":"Program of Thoughts Prompting: Disentangling Computation from Reasoning for Numerical Reasoning Tasks","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2211.12588","snapshot_observed_at":"2026-08-05T10:31:37.759551Z","title":"Program of thoughts prompting: Disentangling computation from reasoning for numerical reasoning tasks","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.759551Z"},"links":{"cited_paper":"/paper/2211.12588","citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:a992a3d7508ee2a9b45a4446e095988447d1eac25155e7b937da6716b477d9d7","observation_id":"1d8dd784-010b-452d-ac80-2bda6dbdfb2a","resolution":{"observed_at":"2026-08-05T10:31:37.759551Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.04548","last_updated":"2025-03-06T15:34:27Z","snapshot_observed_at":"2026-08-07T17:24:22.925386Z","submitted_at":"2025-03-06T15:34:27Z","title":"An Empirical Study on Eliciting and Improving R1-like Reasoning Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.04548","snapshot_observed_at":"2026-08-05T10:31:37.763822Z","title":"An empirical study on eliciting and improving r1-like reasoning models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.763822Z"},"links":{"cited_paper":"/paper/2503.04548","citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:3e71763bb414a3240569de8d6f4887158dac89bea23589a335139fc7c8d90d34","observation_id":"0f1e3433-c858-49cf-852f-f44ed4fbd56d","resolution":{"observed_at":"2026-08-05T10:31:37.763822Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-07T01:45:38.840969Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-05T10:31:37.768029Z","title":"Training verifiers to solve math word problems","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.768029Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:c1928e765b1be5ff53a2e5f0f1fbf05d1c8bc8e74b35fde9d1f3188f477df1c2","observation_id":"0469f239-9c70-4708-bb4d-0844b17ddcdc","resolution":{"observed_at":"2026-08-05T10:31:37.768029Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-05T10:31:37.772049Z","title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.772049Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:3a1d30ef3175699cfa4d273ede05c44ed7ca03b0858f69cc0d3e94a307c1bcfa","observation_id":"022a69a8-5b54-430b-bcf8-f3e022ec8531","resolution":{"observed_at":"2026-08-05T10:31:37.772049Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.01720","last_updated":"2025-02-06T06:42:17Z","snapshot_observed_at":"2026-08-09T14:34:18.630691Z","submitted_at":"2024-10-02T16:32:05Z","title":"Towards a Theoretical Understanding of Synthetic Data in LLM Post-Training: A Reverse-Bottleneck Perspective","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.01720","snapshot_observed_at":"2026-08-05T10:31:37.775535Z","title":"Towards a theoretical understanding of synthetic data in llm post-training: A reverse-bottleneck perspective","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.775535Z"},"links":{"cited_paper":"/paper/2410.01720","citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:79c4e36034ed584e58547dd51cdd3d669e95a5b18f6255193b9c9f2ac7e3746d","observation_id":"8b0b2129-bb5a-4ea7-92e5-788a2ffe8717","resolution":{"observed_at":"2026-08-05T10:31:37.775535Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.15602","last_updated":"2025-06-19T07:05:12Z","snapshot_observed_at":"2026-07-06T20:26:21.639525Z","submitted_at":"2025-01-26T17:05:16Z","title":"Rethinking External Slow-Thinking: From Snowball Errors to Probability of Correct Reasoning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.15602","snapshot_observed_at":"2026-08-05T10:31:37.779402Z","title":"Rethinking external slow-thinking: From snowball errors to probability of correct reasoning, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.779402Z"},"links":{"cited_paper":"/paper/2501.15602","citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:fc5efca1c884091c92c545f5e2a17090897df6c8741c332372cb4a624e41a134","observation_id":"fa980435-17e7-40da-a5ea-0905e43f7941","resolution":{"observed_at":"2026-08-05T10:31:37.779402Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:31:38.532649Z","title":"On distances in uniformly random networks","venue":null,"work_id":"ef97c460-54d9-44ad-8a06-cbb24e129e08","year":2005},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.783759Z"},"links":{"citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:15ea392fe0100b23231915c63e7ac60393b2961ed8e081cf0c24b14bcae9fe3d","observation_id":"9bec0efd-2831-4438-9c99-49d7773451f0","resolution":{"observed_at":"2026-08-05T10:31:38.536573Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.03874","last_updated":"2021-11-08T21:30:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-03-05T18:59:39Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2103.03874","snapshot_observed_at":"2026-08-05T10:31:37.787916Z","title":"Measuring mathematical problem solving with the math dataset","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.787916Z"},"links":{"cited_paper":"/paper/2103.03874","citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:aa129c4bff45e36019bf90bbd82a053401071eb29ce7e590cb0873c233fe5887","observation_id":"758aa007-4123-474a-a24a-2b735e7b363f","resolution":{"observed_at":"2026-08-05T10:31:37.787916Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.03262","last_updated":"2025-11-10T15:11:13Z","snapshot_observed_at":"2026-08-02T05:27:47.490711Z","submitted_at":"2025-01-04T02:08:06Z","title":"REINFORCE++: Stabilizing Critic-Free Policy Optimization with Global Advantage Normalization","version":9},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.03262","snapshot_observed_at":"2026-08-05T10:31:37.792123Z","title":"Reinforce++: An efficient rlhf algorithm with robustness to both prompt and reward models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.792123Z"},"links":{"cited_paper":"/paper/2501.03262","citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:e24a18084e9cee1675062296803dd44dfac36cafa082efdf10b052e135f89cc9","observation_id":"6dd9e2e9-cabd-40dd-9a55-44c83f72e0ef","resolution":{"observed_at":"2026-08-05T10:31:37.792123Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:31:38.520997Z","title":"Survey of hallucination in natural language generation","venue":null,"work_id":"3e15dc50-5aee-4586-b3cf-7aa965bc63f8","year":2023},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.796260Z"},"links":{"citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:dd0810fad2054b1255c4af6b84a47b3f0ff1e588736ec816ec6062ab84c34684","observation_id":"295b6ad4-dade-4ed1-8e1a-2143e6727bdb","resolution":{"observed_at":"2026-08-05T10:31:38.525058Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.11694","last_updated":"2024-12-31T01:38:12Z","snapshot_observed_at":"2026-08-09T03:55:09.041328Z","submitted_at":"2024-11-18T16:15:17Z","title":"Enhancing LLM Reasoning with Reward-guided Tree Search","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.11694","snapshot_observed_at":"2026-08-05T10:31:37.800280Z","title":"Enhancing llm reasoning with reward-guided tree search","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.800280Z"},"links":{"cited_paper":"/paper/2411.11694","citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:889f8a328854ffa3f492957b1633bd4aed458679db2522218d61b5b0aa1133de","observation_id":"0a8439d8-5404-4d31-926f-f181fcc433c4","resolution":{"observed_at":"2026-08-05T10:31:37.800280Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1609.04836","last_updated":"2017-02-09T20:38:16Z","snapshot_observed_at":"2026-07-06T05:10:58.923264Z","submitted_at":"2016-09-15T20:03:06Z","title":"On Large-Batch Training for Deep Learning: Generalization Gap and Sharp Minima","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1609.04836","snapshot_observed_at":"2026-08-05T10:31:37.804582Z","title":"On large-batch training for deep learning: Generalization gap and sharp minima","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.804582Z"},"links":{"cited_paper":"/paper/1609.04836","citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:c3bc98d13916459161798efd071014084f0ae2c7ffc3728590dbe8996d964d2a","observation_id":"ada20603-706a-4c55-be5c-3159391cb56d","resolution":{"observed_at":"2026-08-05T10:31:37.804582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1711.05101","last_updated":"2019-01-04T21:01:49Z","snapshot_observed_at":"2026-08-09T20:34:52.923500Z","submitted_at":"2017-11-14T14:24:06Z","title":"Decoupled Weight Decay Regularization","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1711.05101","snapshot_observed_at":"2026-08-05T10:31:37.808968Z","title":"Decoupled weight decay regularization","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.808968Z"},"links":{"cited_paper":"/paper/1711.05101","citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:f2e12e22bd318e30e75e752f73e8c024d0fd2eb961137aa5e1728d6d3c70be71","observation_id":"0270234c-1444-4e64-a0a9-8a94c501a99b","resolution":{"observed_at":"2026-08-05T10:31:37.808968Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:31:38.510386Z","title":"Some pac-bayesian theorems","venue":null,"work_id":"a8eba528-9707-443e-847a-8886ea844894","year":1998},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.813305Z"},"links":{"citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:f1399034f35171e9f8f288aafbb3469f4ce9e9fc4e4eb921fbbc40a963fefd17","observation_id":"cc6366ea-b12b-4c5c-807e-a77ef077744a","resolution":{"observed_at":"2026-08-05T10:31:38.514186Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-05T10:31:37.819065Z","title":"The llama 3 herd of models, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.819065Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:ab149f7513477d3112a6678e39a9f7b403af62c685a57bec77f43a9cc9ff4ec7","observation_id":"a7d29796-62ed-4555-b21d-81aadb7bb521","resolution":{"observed_at":"2026-08-05T10:31:37.819065Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:31:38.498266Z","title":"Nearest neighbor distance in three-dimensional space","venue":null,"work_id":"eeb39885-df3a-4b2c-9079-bf115fbd9e32","year":2018},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.823520Z"},"links":{"citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:30d576712e268477de656c434cabfb8c776eb9b1fffb9d742861fcd366702501","observation_id":"99b43959-d9de-4c31-8bb8-631a3602bb7c","resolution":{"observed_at":"2026-08-05T10:31:38.502802Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:31:38.486167Z","title":"Learning to reason with llms, 2024","venue":null,"work_id":"2f9e79ab-65e3-4c55-a129-4b373d3639c1","year":2024},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.827688Z"},"links":{"citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:9e4ae4fdfed941b286f82314eed2dec2f1d483e213e62e0b33afc44feda15916","observation_id":"18994fd2-58ee-43b0-8eaf-e72069020449","resolution":{"observed_at":"2026-08-05T10:31:38.490571Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:31:38.472599Z","title":"Introducing openai o3 and o4-mini, 2025","venue":null,"work_id":"661bedd8-bece-466b-aec2-96e083ed1cef","year":2025},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.831809Z"},"links":{"citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:ea12b5fcf6ba000ad42c350f535bff98206522a35a144b5da9844b0d4f71aaf6","observation_id":"a947c17b-463c-415e-bf3c-2bae4b76115f","resolution":{"observed_at":"2026-08-05T10:31:38.477629Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:31:38.459198Z","title":"Qwq: Reflect deeply on the boundaries of the unknown, November 2024","venue":null,"work_id":"f109a8a1-6006-4622-b6c7-ec9ecd581879","year":2024},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.836030Z"},"links":{"citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:6befe3abe14f830c183dcdea84d29b625078619f69e1207b5557ac86f468f358","observation_id":"a54cd083-fb2f-4c11-8b4a-c49510dbb830","resolution":{"observed_at":"2026-08-05T10:31:38.464167Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-05T10:31:37.840224Z","title":"Qwen2.5 technical report, 2025 a","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.840224Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:1393320cefa0862b58891614a62603b66aba90533d45befb92250cffda6b0e53","observation_id":"6e0006be-46fd-4878-b922-e522d5a14d00","resolution":{"observed_at":"2026-08-05T10:31:37.840224Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-08-05T10:31:37.844935Z","title":"Qwen3 technical report, 2025 b","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.844935Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:db0f9bf06f969928b6832a550f210674d7d03aa231e75e63ee0e98d46bcbcffa","observation_id":"f30b8c27-7c1d-4049-a9f9-9ec2465ce29f","resolution":{"observed_at":"2026-08-05T10:31:37.844935Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:31:38.446359Z","title":"Benchmarking prompt sensitivity in large language models","venue":null,"work_id":"e25f29c2-58ea-4ca1-b9cb-53f2a8de9f93","year":2025},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.849362Z"},"links":{"citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:e57c9bbab88a489e99c3debe184d16a25ade215aa4e775dd7d068e905a3e6fdb","observation_id":"ed7d4204-0de5-4b55-ae19-7960911b9773","resolution":{"observed_at":"2026-08-05T10:31:38.451294Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:31:38.433420Z","title":"How much does your data exploration overfit? controlling bias via information usage","venue":null,"work_id":"20d0636a-ef24-4fac-a19c-4662de0fc837","year":2019},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.854081Z"},"links":{"citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:3648ae8e69bde04d77cccad984e9d547ebad5aac95bc2113bd8c48f8a5cb78d1","observation_id":"2571ea31-6b53-4441-9afc-b0e244fadcb5","resolution":{"observed_at":"2026-08-05T10:31:38.438143Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-08-05T10:31:37.858445Z","title":"Proximal policy optimization algorithms","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.858445Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:159d8af1e6326bfa721f4b7ccdb3b92b59233fa5aa01be9c00dfa1a8e6a00d4f","observation_id":"e3cad880-d189-4e67-bcf5-def90fd559b8","resolution":{"observed_at":"2026-08-05T10:31:37.858445Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.12118","last_updated":"2025-02-18T18:54:12Z","snapshot_observed_at":"2026-08-07T18:11:25.040275Z","submitted_at":"2025-02-17T18:43:24Z","title":"Scaling Test-Time Compute Without Verification or RL is Suboptimal","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.12118","snapshot_observed_at":"2026-08-05T10:31:37.862981Z","title":"Scaling test-time compute without verification or rl is suboptimal","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.862981Z"},"links":{"cited_paper":"/paper/2502.12118","citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:30f09d0ebfd0ca19ccbf2fb7ea0597871501da7fec6681ad1b46b46333b645d0","observation_id":"c3121a39-00b2-46a5-a1ab-654f921f914f","resolution":{"observed_at":"2026-08-05T10:31:37.862981Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.10947","last_updated":"2026-02-25T01:06:05Z","snapshot_observed_at":"2026-07-30T09:54:40.100382Z","submitted_at":"2025-06-12T17:49:55Z","title":"Spurious Rewards: Rethinking Training Signals in RLVR","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.10947","snapshot_observed_at":"2026-08-05T10:31:37.869115Z","title":"Spurious rewards: Rethinking training signals in rlvr","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.869115Z"},"links":{"cited_paper":"/paper/2506.10947","citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:5016e089bcd593bba2a534f17f8180d7153a95346a16d635ca9f3e466647d475","observation_id":"70d53b55-92e9-418f-8b2e-2d0617b2bff0","resolution":{"observed_at":"2026-08-05T10:31:37.869115Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-05T10:31:37.873877Z","title":"Deepseekmath: Pushing the limits of mathematical reasoning in open language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.873877Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:00e52ecb5041590111b2001a95dce37e9bc39c928cf8927756e13467592f122c","observation_id":"ddd1d032-1732-4740-a8a4-bf1983309fac","resolution":{"observed_at":"2026-08-05T10:31:37.873877Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:31:37.879593Z","title":"A bayesian perspective on generalization and stochastic gradient descent","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.879593Z"},"links":{"citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:4d366ce9b6708721c1a559f616d46bc3bc3bae253e27984a8f3ef0de3d75b6da","observation_id":"a67f407a-8b36-4419-adf6-781640bcc753","resolution":{"observed_at":"2026-08-05T10:31:37.879593Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03314","last_updated":"2024-08-06T17:35:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:35:05Z","title":"Scaling LLM Test-Time Compute Optimally can be More Effective than Scaling Model Parameters","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03314","snapshot_observed_at":"2026-08-05T10:31:37.884342Z","title":"Scaling llm test-time compute optimally can be more effective than scaling model parameters","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.884342Z"},"links":{"cited_paper":"/paper/2408.03314","citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:8d3c32ec54a42c317e82b4aa3156446218ebacb51d67eeb33514bcf63ed3ce9f","observation_id":"21dc2f48-eff6-498b-8b74-69cc9287bd3b","resolution":{"observed_at":"2026-08-05T10:31:37.884342Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.16419","last_updated":"2025-08-21T19:14:40Z","snapshot_observed_at":"2026-08-07T04:27:23.738927Z","submitted_at":"2025-03-20T17:59:38Z","title":"Stop Overthinking: A Survey on Efficient Reasoning for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.16419","snapshot_observed_at":"2026-08-05T10:31:37.889298Z","title":"Stop overthinking: A survey on efficient reasoning for large language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.889298Z"},"links":{"cited_paper":"/paper/2503.16419","citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:46fe77c26fe15d3893a1e9ae1c386786e757c3fe2101b76e08f0e835855c95bc","observation_id":"b671102f-5d38-4cf1-801e-336e7c3bd280","resolution":{"observed_at":"2026-08-05T10:31:37.889298Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.11984","last_updated":"2025-07-10T14:18:01Z","snapshot_observed_at":"2026-08-03T11:16:52.526955Z","submitted_at":"2024-11-18T19:14:36Z","title":"Understanding Chain-of-Thought in LLMs through Information Theory","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.11984","snapshot_observed_at":"2026-08-05T10:31:37.894683Z","title":"Understanding chain-of-thought in llms through information theory","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.894683Z"},"links":{"cited_paper":"/paper/2411.11984","citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:1ec476f8273b8ea227c4a619889b5e88aabd6d3678a562a06ef15f35cc01ff5c","observation_id":"b5bd238d-19cf-4f4f-afa0-29c1ecc4ad53","resolution":{"observed_at":"2026-08-05T10:31:37.894683Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:31:38.413176Z","title":"Alphazero-like tree-search can guide large language model decoding and training","venue":null,"work_id":"306e61ce-6355-4b2b-9845-47522a832edd","year":2024},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.899762Z"},"links":{"citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:12ef0ac3b3953118f20905df37baa739bf6bebf7f609fe25151ed2282ea5cb89","observation_id":"fcb68915-e810-4f07-9ce4-755df89ed3d5","resolution":{"observed_at":"2026-08-05T10:31:38.418602Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.11171","last_updated":"2023-03-07T17:57:37Z","snapshot_observed_at":"2026-07-06T12:50:22.773056Z","submitted_at":"2022-03-21T17:48:52Z","title":"Self-Consistency Improves Chain of Thought Reasoning in Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.11171","snapshot_observed_at":"2026-08-05T10:31:37.904517Z","title":"Self-consistency improves chain of thought reasoning in language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.904517Z"},"links":{"cited_paper":"/paper/2203.11171","citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:83ba628b3f40a6612b9aca83c3a5a0b43cf1374b7caf998db0cbd16bf139a128","observation_id":"f9dfb13e-8c84-4749-abd2-807892557a2e","resolution":{"observed_at":"2026-08-05T10:31:37.904517Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.07682","last_updated":"2022-10-26T05:06:24Z","snapshot_observed_at":"2026-08-02T15:56:35.249569Z","submitted_at":"2022-06-15T17:32:01Z","title":"Emergent Abilities of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.07682","snapshot_observed_at":"2026-08-05T10:31:37.909588Z","title":"Emergent abilities of large language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.909588Z"},"links":{"cited_paper":"/paper/2206.07682","citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:398ee1c27cd9792fd254c539893d11354d99c5baa059b33c157ccec7da205581","observation_id":"4f90a21e-332b-47fe-8db8-77670c608173","resolution":{"observed_at":"2026-08-05T10:31:37.909588Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:31:37.914448Z","title":"Chain-of-thought prompting elicits reasoning in large language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.914448Z"},"links":{"citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:bbf2e8e84b637acddf986fbce6d95af45351c39b00308cdba0c606510af7f5e3","observation_id":"5f0d9a22-a300-49fa-b531-53970135c7ae","resolution":{"observed_at":"2026-08-05T10:31:37.914448Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00724","last_updated":"2025-03-03T07:53:32Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-01T17:16:04Z","title":"Inference Scaling Laws: An Empirical Analysis of Compute-Optimal Inference for Problem-Solving with Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00724","snapshot_observed_at":"2026-08-05T10:31:37.919359Z","title":"Inference scaling laws: An empirical analysis of compute-optimal inference for problem-solving with language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.919359Z"},"links":{"cited_paper":"/paper/2408.00724","citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:7d810c1fa6aaec4b1e951e0f4f6f05087c32f048fd672d0f5723540764c71370","observation_id":"d155c5de-8c81-4195-a9f0-170c5792b7e6","resolution":{"observed_at":"2026-08-05T10:31:37.919359Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:31:38.391429Z","title":"Information-theoretic analysis of generalization capability of learning algorithms","venue":null,"work_id":"ccafacff-69bb-4bfb-a0d5-2b1592d04f62","year":2017},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.924011Z"},"links":{"citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:487dc4224e3e9a96f3e26f6831f11475e5c428f4381e6d67295dbb524eb873af","observation_id":"295e2ff3-2aef-472a-9343-23a1c135d222","resolution":{"observed_at":"2026-08-05T10:31:38.398678Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:31:37.928512Z","title":"Tree of thoughts: Deliberate problem solving with large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.928512Z"},"links":{"citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:afdf6bb949e39a887a01a18e38efb90603b94b2b5fddd863afe9e091413d3566","observation_id":"f010506b-53ab-4da0-8a04-855bf8c5800b","resolution":{"observed_at":"2026-08-05T10:31:37.928512Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.14476","last_updated":"2025-05-20T01:37:34Z","snapshot_observed_at":"2026-08-02T01:40:54.187278Z","submitted_at":"2025-03-18T17:49:06Z","title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.14476","snapshot_observed_at":"2026-08-05T10:31:37.933115Z","title":"Dapo: An open-source llm reinforcement learning system at scale","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.933115Z"},"links":{"cited_paper":"/paper/2503.14476","citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:0a17ff84e253ac52d914347ddbb410a85aa33af3ce634bb0cbc4c1eef5bceba0","observation_id":"25fff4de-bd51-413e-97d0-775fd7de3d63","resolution":{"observed_at":"2026-08-05T10:31:37.933115Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-07T17:33:49.286933Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-05T10:31:37.938067Z","title":"What's behind ppo's collapse in long-cot? value optimization holds the secret","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.938067Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:6434fe9ecb6b8a09a94b1eb189b066fe4dc717918b65320a3f39819185882653","observation_id":"5c9667f0-812e-4e67-8d41-0e8f7003c117","resolution":{"observed_at":"2026-08-05T10:31:37.938067Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.13837","last_updated":"2025-11-24T06:11:04Z","snapshot_observed_at":"2026-07-06T21:11:34.701779Z","submitted_at":"2025-04-18T17:59:56Z","title":"Does Reinforcement Learning Really Incentivize Reasoning Capacity in LLMs Beyond the Base Model?","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.13837","snapshot_observed_at":"2026-08-05T10:31:37.942705Z","title":"Does reinforcement learning really incentivize reasoning capacity in llms beyond the base model? arXiv preprint arXiv:2504.13837, 2025 a","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.942705Z"},"links":{"cited_paper":"/paper/2504.13837","citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:c112b53a2cdf0264c62f591418801810bfca9c7ed2035520c8842a97d715fbed","observation_id":"c7f1001c-2bab-4f60-8465-b46f3d0069de","resolution":{"observed_at":"2026-08-05T10:31:37.942705Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.05118","last_updated":"2025-04-11T02:54:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-07T14:21:11Z","title":"VAPO: Efficient and Reliable Reinforcement Learning for Advanced Reasoning Tasks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.05118","snapshot_observed_at":"2026-08-05T10:31:37.947268Z","title":"Vapo: Efficient and reliable reinforcement learning for advanced reasoning tasks","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.947268Z"},"links":{"cited_paper":"/paper/2504.05118","citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:946194458b0f69d5c0f031c39e4ef0695db9e610cd371f5a93099d5faac7a251","observation_id":"97bdd282-cc34-4964-b295-0a99b8da6273","resolution":{"observed_at":"2026-08-05T10:31:37.947268Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.09629","last_updated":"2024-03-18T07:56:48Z","snapshot_observed_at":"2026-07-31T09:25:34.205362Z","submitted_at":"2024-03-14T17:58:16Z","title":"Quiet-STaR: Language Models Can Teach Themselves to Think Before Speaking","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.09629","snapshot_observed_at":"2026-08-05T10:31:37.951774Z","title":"Quiet-star: Language models can teach themselves to think before speaking","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.951774Z"},"links":{"cited_paper":"/paper/2403.09629","citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:d17859164b6d89c53e2df62a62dccf2867453f717a4751bd0c9d6fbe61f1ea15","observation_id":"1d10c2c0-d047-4c30-b3ad-a1a73d9121db","resolution":{"observed_at":"2026-08-05T10:31:37.951774Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:31:37.956608Z","title":"Rest-mcts*: Llm self-training via process reward guided tree search","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.956608Z"},"links":{"citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:b93796b08afcab3a5bb05f5cb0161be1959a15bc5c998dec04bbdb68ebf9fbb7","observation_id":"6174444d-a37b-45bc-95a6-2838c75b832c","resolution":{"observed_at":"2026-08-05T10:31:37.956608Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.07912","last_updated":"2025-08-07T23:50:47Z","snapshot_observed_at":"2026-08-09T01:40:08.645048Z","submitted_at":"2025-04-10T17:15:53Z","title":"Echo Chamber: RL Post-training Amplifies Behaviors Learned in Pretraining","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.07912","snapshot_observed_at":"2026-08-05T10:31:37.961065Z","title":"Echo chamber: Rl post-training amplifies behaviors learned in pretraining","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.961065Z"},"links":{"cited_paper":"/paper/2504.07912","citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:c5b974c0c87ea6601d12b8a1871a344e98388cb270338dc4b2ec23b38b53632d","observation_id":"cbe4cbfe-156e-46e5-874d-4a5c2542018f","resolution":{"observed_at":"2026-08-05T10:31:37.961065Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:31:38.359932Z","title":"Prosa: Assessing and understanding the prompt sensitivity of llms","venue":null,"work_id":"02e08d9d-fd58-48de-ad8e-29709f1b2afd","year":2024},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.966101Z"},"links":{"citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:caf2bd412caf3a1d174dfb005ff5b63a5a166ff3a8ff1d37a08f8b32aa10f499","observation_id":"238330f5-2f6a-4340-aa8c-b4321bac6f44","resolution":{"observed_at":"2026-08-05T10:31:38.367597Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:31:37.971079Z","title":"@esa (Ref","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.971079Z"},"links":{"citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:450b3f2ba433ada62b1179bcecb166581518a06f5e9eaaf4981859be7c6b995c","observation_id":"fa14debb-98ee-4397-88ac-456b940528a8","resolution":{"observed_at":"2026-08-05T10:31:37.971079Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:31:37.976071Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.976071Z"},"links":{"citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:ae5bfaed42213d4c80df7d6cd1420879b739378dae851783674cebc18efcb076","observation_id":"9c1cc768-96eb-40c5-9962-4f4079d4f000","resolution":{"observed_at":"2026-08-05T10:31:37.976071Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T10:31:37.980968Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.980968Z"},"links":{"citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:88b5da5ddc263db0097d398133ee98b9bb6f52d192834b79e76e6021f4b27135","observation_id":"2db63415-3ea4-479d-9bbd-00fd28ff3b2b","resolution":{"observed_at":"2026-08-05T10:31:37.980968Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","latest_version":4,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-09T08:53:17.340894Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought"},"reference_resolution":{"displayed":53,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":41,"verified_exact":0,"verified_fuzzy":12},"total_outbound_references":53},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 53 of 53 outbound references and 1 inbound Pith citation observation for arXiv:2509.04027."}