{"as_of":"2026-08-16T23:08:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a5e9832eaf3d9d183676ba4e219bdcfe0e14727fbe1aa2dd2c7f21247f53cc0d","coverage":[{"denominator":31,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":31,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-01T06:17:30.923859Z","state":"measured"},{"denominator":31,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":31,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-16T06:30:59.297886+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2607.21971/citation-record","integrity":"/paper/2607.21971/integrity","json":"/paper/2607.21971/citation-record.json","paper":"/paper/2607.21971"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T06:17:27.893513Z","title":"Tool-r0: Self-evolving llm agents for tool-learning from zero data.arXiv preprint arXiv:2602.21320,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21971","last_updated":"2026-07-24T04:35:29Z","snapshot_observed_at":"2026-08-15T08:18:09.144189Z","submitted_at":"2026-07-24T04:35:29Z","title":"Teaching LLMs to Self-Evolve: Cultivating Core Meta-Skills with Reinforcement Learning","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-01T06:17:27.893513Z"},"links":{"citing_paper":"/paper/2607.21971"},"observation_digest":"sha256:471c93cf0698a942462bd0122f927f14ffb257b033f3b4f21576b4d8b54554ce","observation_id":"53922a82-2efd-40f5-bf5b-a932d4c9fb47","resolution":{"observed_at":"2026-08-01T06:17:27.893513Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-08-15T12:33:55.451951Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-01T06:17:28.286504Z","title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning.arXiv preprint arXiv:2501.12948, 2025a","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21971","last_updated":"2026-07-24T04:35:29Z","snapshot_observed_at":"2026-08-15T08:18:09.144189Z","submitted_at":"2026-07-24T04:35:29Z","title":"Teaching LLMs to Self-Evolve: Cultivating Core Meta-Skills with Reinforcement Learning","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-01T06:17:28.286504Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2607.21971"},"observation_digest":"sha256:37e4cfd05809ffd6d92d2d63f8519055cd9a507003d4e29f045c8c8d6fc89d54","observation_id":"e454a080-4ae4-404d-a318-2b9023df8b0f","resolution":{"observed_at":"2026-08-01T06:17:28.286504Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.14509","last_updated":"2023-10-04T16:51:13Z","snapshot_observed_at":"2026-08-13T22:49:56.824755Z","submitted_at":"2023-09-25T20:15:57Z","title":"DeepSpeed Ulysses: System Optimizations for Enabling Training of Extreme Long Sequence Transformer Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.14509","snapshot_observed_at":"2026-08-01T06:17:28.461299Z","title":"Deepspeed ulysses: System optimiza- tions for enabling training of extreme long sequence transformer models.arXiv preprint arXiv:2309.14509,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21971","last_updated":"2026-07-24T04:35:29Z","snapshot_observed_at":"2026-08-15T08:18:09.144189Z","submitted_at":"2026-07-24T04:35:29Z","title":"Teaching LLMs to Self-Evolve: Cultivating Core Meta-Skills with Reinforcement Learning","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-01T06:17:28.461299Z"},"links":{"cited_paper":"/paper/2309.14509","citing_paper":"/paper/2607.21971"},"observation_digest":"sha256:de75c9c5afe76286e1f0aa8df53655ca92b9a785cf9d004b11240fa5ca6f8641","observation_id":"b6fa36e3-4de7-41e5-a24e-a19dce39c1f2","resolution":{"observed_at":"2026-08-01T06:17:28.461299Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.14852","last_updated":"2023-12-27T10:09:18Z","snapshot_observed_at":"2026-08-16T14:32:08.450873Z","submitted_at":"2023-12-22T17:25:42Z","title":"TACO: Topics in Algorithmic COde generation dataset","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.14852","snapshot_observed_at":"2026-08-01T06:17:28.636771Z","title":"Taco: Topics in algorithmic code generation dataset.arXiv preprint arXiv:2312.14852,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21971","last_updated":"2026-07-24T04:35:29Z","snapshot_observed_at":"2026-08-15T08:18:09.144189Z","submitted_at":"2026-07-24T04:35:29Z","title":"Teaching LLMs to Self-Evolve: Cultivating Core Meta-Skills with Reinforcement Learning","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-01T06:17:28.636771Z"},"links":{"cited_paper":"/paper/2312.14852","citing_paper":"/paper/2607.21971"},"observation_digest":"sha256:24958b23d26d68210c0a15e9cc8c414576a537ac3a1f38d2c075d5d45ae52c7a","observation_id":"000030d2-0b33-4d6f-aa88-676fcb5bbf4e","resolution":{"observed_at":"2026-08-01T06:17:28.636771Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T06:17:28.698988Z","title":"s1: Simple test-time scaling","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.21971","last_updated":"2026-07-24T04:35:29Z","snapshot_observed_at":"2026-08-15T08:18:09.144189Z","submitted_at":"2026-07-24T04:35:29Z","title":"Teaching LLMs to Self-Evolve: Cultivating Core Meta-Skills with Reinforcement Learning","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-01T06:17:28.698988Z"},"links":{"citing_paper":"/paper/2607.21971"},"observation_digest":"sha256:de944525322ab7965c22b5784307e78d710c84bacbeca9029d073fb5633899c7","observation_id":"435da973-f07f-41fe-a144-ae5e61668802","resolution":{"observed_at":"2026-08-01T06:17:28.698988Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.13131","last_updated":"2025-06-16T06:37:18Z","snapshot_observed_at":"2026-08-11T23:48:15.512738Z","submitted_at":"2025-06-16T06:37:18Z","title":"AlphaEvolve: A coding agent for scientific and algorithmic discovery","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.13131","snapshot_observed_at":"2026-08-01T06:17:28.779932Z","title":"Alphaevolve: A coding agent for scientific and algorithmic discovery","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21971","last_updated":"2026-07-24T04:35:29Z","snapshot_observed_at":"2026-08-15T08:18:09.144189Z","submitted_at":"2026-07-24T04:35:29Z","title":"Teaching LLMs to Self-Evolve: Cultivating Core Meta-Skills with Reinforcement Learning","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-01T06:17:28.779932Z"},"links":{"cited_paper":"/paper/2506.13131","citing_paper":"/paper/2607.21971"},"observation_digest":"sha256:a36c08b740d1509cb1760931dae787448cf46d1c7596bd3e823d758b386befc5","observation_id":"bfd4648f-a578-497b-910d-7a094c0f71ee","resolution":{"observed_at":"2026-08-01T06:17:28.779932Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T06:17:28.952018Z","title":"Rafael Rafailov, Archit Sharma, Eric Mitchell, Christopher D Manning, Stefano Ermon, and Chelsea Finn","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21971","last_updated":"2026-07-24T04:35:29Z","snapshot_observed_at":"2026-08-15T08:18:09.144189Z","submitted_at":"2026-07-24T04:35:29Z","title":"Teaching LLMs to Self-Evolve: Cultivating Core Meta-Skills with Reinforcement Learning","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-01T06:17:28.952018Z"},"links":{"citing_paper":"/paper/2607.21971"},"observation_digest":"sha256:6eb78ff927d2396b9cadb80180d6599826c741e257b2254080f8b80427fea92f","observation_id":"8d736a13-c85b-4f74-8951-abb03b108e4e","resolution":{"observed_at":"2026-08-01T06:17:28.952018Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-01T06:17:29.146347Z","title":"DeepSeekMath: Pushing the limits of mathematical reasoning in open language models.arXiv preprint arXiv:2402.03300,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21971","last_updated":"2026-07-24T04:35:29Z","snapshot_observed_at":"2026-08-15T08:18:09.144189Z","submitted_at":"2026-07-24T04:35:29Z","title":"Teaching LLMs to Self-Evolve: Cultivating Core Meta-Skills with Reinforcement Learning","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-01T06:17:29.146347Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2607.21971"},"observation_digest":"sha256:103a2ebd242ef490caaec5b917f14a29e17cb54be16b619aef5b6997295a89d1","observation_id":"e1f3796a-0c3e-46af-b599-7674d5e75dbe","resolution":{"observed_at":"2026-08-01T06:17:29.146347Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.19256","last_updated":"2024-10-02T04:01:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-28T06:20:03Z","title":"HybridFlow: A Flexible and Efficient RLHF Framework","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.19256","snapshot_observed_at":"2026-08-01T06:17:29.227523Z","title":"Guangming Sheng, Chi Zhang, Zilingfeng Ye, Xibin Wu, Wang Zhang, Ru Zhang, Yanghua Peng, Haibin Lin, and Chuan Wu","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21971","last_updated":"2026-07-24T04:35:29Z","snapshot_observed_at":"2026-08-15T08:18:09.144189Z","submitted_at":"2026-07-24T04:35:29Z","title":"Teaching LLMs to Self-Evolve: Cultivating Core Meta-Skills with Reinforcement Learning","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-01T06:17:29.227523Z"},"links":{"cited_paper":"/paper/2409.19256","citing_paper":"/paper/2607.21971"},"observation_digest":"sha256:b882e933ec78d41efb7d4606ca9bd9a552a42359f45de15b07a2f7e623b32003","observation_id":"173508c7-8ccc-400f-8a0e-1c8cfbaf5784","resolution":{"observed_at":"2026-08-01T06:17:29.227523Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.10952","last_updated":"2024-04-16T23:27:38Z","snapshot_observed_at":"2026-08-16T14:00:18.913865Z","submitted_at":"2024-04-16T23:27:38Z","title":"Can Language Models Solve Olympiad Programming?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.10952","snapshot_observed_at":"2026-08-01T06:17:29.317498Z","title":"Can language models solve olympiad programming?arXiv preprint arXiv:2404.10952,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21971","last_updated":"2026-07-24T04:35:29Z","snapshot_observed_at":"2026-08-15T08:18:09.144189Z","submitted_at":"2026-07-24T04:35:29Z","title":"Teaching LLMs to Self-Evolve: Cultivating Core Meta-Skills with Reinforcement Learning","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-01T06:17:29.317498Z"},"links":{"cited_paper":"/paper/2404.10952","citing_paper":"/paper/2607.21971"},"observation_digest":"sha256:aa17f3f21a2442c8e9fcbccb826efefb1baff1ab16afe8d479393a3ee44f1f12","observation_id":"6538a143-6205-42b7-b0ea-e1734f4021e0","resolution":{"observed_at":"2026-08-01T06:17:29.317498Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04109","last_updated":"2024-09-06T08:25:03Z","snapshot_observed_at":"2026-08-16T13:20:49.991369Z","submitted_at":"2024-09-06T08:25:03Z","title":"Can LLMs Generate Novel Research Ideas? A Large-Scale Human Study with 100+ NLP Researchers","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.04109","snapshot_observed_at":"2026-08-01T06:17:29.411352Z","title":"Can llms generate novel research ideas? a large-scale human study with 100+ nlp researchers.arXiv preprint arXiv:2409.04109,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21971","last_updated":"2026-07-24T04:35:29Z","snapshot_observed_at":"2026-08-15T08:18:09.144189Z","submitted_at":"2026-07-24T04:35:29Z","title":"Teaching LLMs to Self-Evolve: Cultivating Core Meta-Skills with Reinforcement Learning","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-01T06:17:29.411352Z"},"links":{"cited_paper":"/paper/2409.04109","citing_paper":"/paper/2607.21971"},"observation_digest":"sha256:9c4d9989d2cec2cf6b99465d06395d7f8f05c1f76af341321469aac5e0b77a3e","observation_id":"ecfe5ef8-f368-4708-a020-1d19cc1f19c8","resolution":{"observed_at":"2026-08-01T06:17:29.411352Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03314","last_updated":"2024-08-06T17:35:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:35:05Z","title":"Scaling LLM Test-Time Compute Optimally can be More Effective than Scaling Model Parameters","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03314","snapshot_observed_at":"2026-08-01T06:17:29.493127Z","title":"Scaling LLM test-time com- pute optimally can be more effective than scaling model parameters.arXiv preprint arXiv:2408.03314,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21971","last_updated":"2026-07-24T04:35:29Z","snapshot_observed_at":"2026-08-15T08:18:09.144189Z","submitted_at":"2026-07-24T04:35:29Z","title":"Teaching LLMs to Self-Evolve: Cultivating Core Meta-Skills with Reinforcement Learning","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-01T06:17:29.493127Z"},"links":{"cited_paper":"/paper/2408.03314","citing_paper":"/paper/2607.21971"},"observation_digest":"sha256:8494a758db8ae92514cfaf8fd69dadb6dded5c167085bc26fe82a357c0bb8193","observation_id":"e9b24a70-7188-4c75-9263-5862a13fbec2","resolution":{"observed_at":"2026-08-01T06:17:29.493127Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T06:17:29.584985Z","title":"Large language model reasoning failures","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21971","last_updated":"2026-07-24T04:35:29Z","snapshot_observed_at":"2026-08-15T08:18:09.144189Z","submitted_at":"2026-07-24T04:35:29Z","title":"Teaching LLMs to Self-Evolve: Cultivating Core Meta-Skills with Reinforcement Learning","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-01T06:17:29.584985Z"},"links":{"citing_paper":"/paper/2607.21971"},"observation_digest":"sha256:4430bbe7b8d940dd87fd4d854074e793282f1775fc72cd4d33e1855fdbc7c126","observation_id":"4669a225-09d9-4622-864f-4cb0d61d9aec","resolution":{"observed_at":"2026-08-01T06:17:29.584985Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T06:17:29.668687Z","title":"Under review","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21971","last_updated":"2026-07-24T04:35:29Z","snapshot_observed_at":"2026-08-15T08:18:09.144189Z","submitted_at":"2026-07-24T04:35:29Z","title":"Teaching LLMs to Self-Evolve: Cultivating Core Meta-Skills with Reinforcement Learning","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-01T06:17:29.668687Z"},"links":{"citing_paper":"/paper/2607.21971"},"observation_digest":"sha256:7f19ca39075ef1b9110d8a35091521fe6552b252c9ca019df8aedef10a46a896","observation_id":"2b6e34b7-b0d0-4270-8d81-542ffb413ae6","resolution":{"observed_at":"2026-08-01T06:17:29.668687Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2511.23473","last_updated":"2025-11-28T18:58:14Z","snapshot_observed_at":"2026-08-04T17:31:45.294646Z","submitted_at":"2025-11-28T18:58:14Z","title":"ThetaEvolve: Test-time Learning on Open Problems","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2511.23473","snapshot_observed_at":"2026-08-01T06:17:29.754644Z","title":"Self-consistency improves chain of thought reasoning in language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21971","last_updated":"2026-07-24T04:35:29Z","snapshot_observed_at":"2026-08-15T08:18:09.144189Z","submitted_at":"2026-07-24T04:35:29Z","title":"Teaching LLMs to Self-Evolve: Cultivating Core Meta-Skills with Reinforcement Learning","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-01T06:17:29.754644Z"},"links":{"cited_paper":"/paper/2511.23473","citing_paper":"/paper/2607.21971"},"observation_digest":"sha256:50c4c6f8499522d84aca497829acede83bb112bfdfbd0e9b2ee121ff57a0400c","observation_id":"d221d477-c10b-4927-91fe-6ba648b9939b","resolution":{"observed_at":"2026-08-01T06:17:29.754644Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.14655","last_updated":"2025-04-20T15:28:16Z","snapshot_observed_at":"2026-08-16T11:41:37.083470Z","submitted_at":"2025-04-20T15:28:16Z","title":"LeetCodeDataset: A Temporal Dataset for Robust Evaluation and Efficient Training of Code LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.14655","snapshot_observed_at":"2026-08-01T06:17:29.782230Z","title":"Leetcodedataset: A temporal dataset for robust evaluation and efficient training of code llms.arXiv preprint arXiv:2504.14655,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21971","last_updated":"2026-07-24T04:35:29Z","snapshot_observed_at":"2026-08-15T08:18:09.144189Z","submitted_at":"2026-07-24T04:35:29Z","title":"Teaching LLMs to Self-Evolve: Cultivating Core Meta-Skills with Reinforcement Learning","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-01T06:17:29.782230Z"},"links":{"cited_paper":"/paper/2504.14655","citing_paper":"/paper/2607.21971"},"observation_digest":"sha256:378ae44782632c9347909cd3dd46f8f117983e79776ba462be1cfc419a47d245","observation_id":"63972755-8e1c-4bd2-b8cb-66c9de8a73bd","resolution":{"observed_at":"2026-08-01T06:17:29.782230Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-08-01T06:17:29.938488Z","title":"Qwen3 technical report.arXiv preprint arXiv:2505.09388,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21971","last_updated":"2026-07-24T04:35:29Z","snapshot_observed_at":"2026-08-15T08:18:09.144189Z","submitted_at":"2026-07-24T04:35:29Z","title":"Teaching LLMs to Self-Evolve: Cultivating Core Meta-Skills with Reinforcement Learning","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-01T06:17:29.938488Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2607.21971"},"observation_digest":"sha256:0cb6d4b9fb9793bcfb94fb694357885b5f2c3b9ca9ce8e781008cc82ca2ff6eb","observation_id":"c443b250-a9f3-4583-95bd-86a5d31f7d72","resolution":{"observed_at":"2026-08-01T06:17:29.938488Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.19446","last_updated":"2024-02-29T18:45:56Z","snapshot_observed_at":"2026-08-16T14:14:00.679899Z","submitted_at":"2024-02-29T18:45:56Z","title":"ArCHer: Training Language Model Agents via Hierarchical Multi-Turn RL","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.19446","snapshot_observed_at":"2026-08-01T06:17:30.221262Z","title":"Archer: Training language model agents via hierarchical multi-turn rl.arXiv preprint arXiv:2402.19446,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21971","last_updated":"2026-07-24T04:35:29Z","snapshot_observed_at":"2026-08-15T08:18:09.144189Z","submitted_at":"2026-07-24T04:35:29Z","title":"Teaching LLMs to Self-Evolve: Cultivating Core Meta-Skills with Reinforcement Learning","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-01T06:17:30.221262Z"},"links":{"cited_paper":"/paper/2402.19446","citing_paper":"/paper/2607.21971"},"observation_digest":"sha256:cb457239c1f6a9d70ccf0f637a7935ca9509e0bcff73dddf8986a234184194bb","observation_id":"1d592169-b77b-4224-8570-132ba46210a4","resolution":{"observed_at":"2026-08-01T06:17:30.221262Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.15478","last_updated":"2025-03-19T17:55:08Z","snapshot_observed_at":"2026-08-16T12:48:30.366139Z","submitted_at":"2025-03-19T17:55:08Z","title":"SWEET-RL: Training Multi-Turn LLM Agents on Collaborative Reasoning Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.15478","snapshot_observed_at":"2026-08-01T06:17:30.431790Z","title":"Sweet-rl: Training multi-turn llm agents on collaborative reasoning tasks","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21971","last_updated":"2026-07-24T04:35:29Z","snapshot_observed_at":"2026-08-15T08:18:09.144189Z","submitted_at":"2026-07-24T04:35:29Z","title":"Teaching LLMs to Self-Evolve: Cultivating Core Meta-Skills with Reinforcement Learning","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-01T06:17:30.431790Z"},"links":{"cited_paper":"/paper/2503.15478","citing_paper":"/paper/2607.21971"},"observation_digest":"sha256:1a75e4647f9fb14dea2d8881e678ccc3e35a58df2fc99de3ea1a957406006950","observation_id":"fb439d78-5096-4625-ab82-618defb3f50a","resolution":{"observed_at":"2026-08-01T06:17:30.431790Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1909.08593","last_updated":"2020-01-08T23:02:36Z","snapshot_observed_at":"2026-08-16T00:15:57.597094Z","submitted_at":"2019-09-18T17:33:39Z","title":"Fine-Tuning Language Models from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.08593","snapshot_observed_at":"2026-08-01T06:17:30.601405Z","title":"Fine-tuning language models from human prefer- ences.arXiv preprint arXiv:1909.08593,","venue":null,"work_id":null,"year":1909},"citing_paper":{"arxiv_id":"2607.21971","last_updated":"2026-07-24T04:35:29Z","snapshot_observed_at":"2026-08-15T08:18:09.144189Z","submitted_at":"2026-07-24T04:35:29Z","title":"Teaching LLMs to Self-Evolve: Cultivating Core Meta-Skills with Reinforcement Learning","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-01T06:17:30.601405Z"},"links":{"cited_paper":"/paper/1909.08593","citing_paper":"/paper/2607.21971"},"observation_digest":"sha256:e176fb52815ce36f24ec81c4117806a173b476dc70cc207372b567e81ef14c1a","observation_id":"25b3f886-63a0-4844-9618-e8ea366327c8","resolution":{"observed_at":"2026-08-01T06:17:30.601405Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T06:17:30.923859Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2607.21971","last_updated":"2026-07-24T04:35:29Z","snapshot_observed_at":"2026-08-15T08:18:09.144189Z","submitted_at":"2026-07-24T04:35:29Z","title":"Teaching LLMs to Self-Evolve: Cultivating Core Meta-Skills with Reinforcement Learning","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-01T06:17:30.923859Z"},"links":{"citing_paper":"/paper/2607.21971"},"observation_digest":"sha256:04910dbd90f43d9d04b50d867e8bf38e304def5ec5831fc4c4f415c69dc6e310","observation_id":"3b0629b3-dfb0-46ce-bba4-bd80c8a03f01","resolution":{"observed_at":"2026-08-01T06:17:30.923859Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-16T01:22:08.293337Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-01T06:17:30.106561Z","title":"doi: 10.1137/0218082","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21971","last_updated":"2026-07-24T04:35:29Z","snapshot_observed_at":"2026-08-15T08:18:09.144189Z","submitted_at":"2026-07-24T04:35:29Z","title":"Teaching LLMs to Self-Evolve: Cultivating Core Meta-Skills with Reinforcement Learning","version":1},"reference_index":1989,"source":"pdf_text","source_observed_at":"2026-08-01T06:17:30.106561Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2607.21971"},"observation_digest":"sha256:d1bf3e64a966c4a075dd5c52782336c0dbcdc1db4b87c37e287d98bdc33dfb9a","observation_id":"f2543b24-09ef-4288-bffd-db2be8b8fe91","resolution":{"observed_at":"2026-08-01T06:17:30.106561Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01456","last_updated":"2025-09-26T09:25:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-03T15:43:48Z","title":"Process Reinforcement through Implicit Rewards","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.01456","snapshot_observed_at":"2026-08-01T06:17:28.092385Z","title":"Under review","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21971","last_updated":"2026-07-24T04:35:29Z","snapshot_observed_at":"2026-08-15T08:18:09.144189Z","submitted_at":"2026-07-24T04:35:29Z","title":"Teaching LLMs to Self-Evolve: Cultivating Core Meta-Skills with Reinforcement Learning","version":1},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-01T06:17:28.092385Z"},"links":{"cited_paper":"/paper/2502.01456","citing_paper":"/paper/2607.21971"},"observation_digest":"sha256:40658c4b0aabc43e9fab2a41d53e3785d793e8c79e83cf82301bb7daa0edbc1c","observation_id":"0690d2b4-9e15-4ce8-b129-4e2ec6d75196","resolution":{"observed_at":"2026-08-01T06:17:28.092385Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T06:17:30.720072Z","title":"Under review","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.21971","last_updated":"2026-07-24T04:35:29Z","snapshot_observed_at":"2026-08-15T08:18:09.144189Z","submitted_at":"2026-07-24T04:35:29Z","title":"Teaching LLMs to Self-Evolve: Cultivating Core Meta-Skills with Reinforcement Learning","version":1},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-01T06:17:30.720072Z"},"links":{"citing_paper":"/paper/2607.21971"},"observation_digest":"sha256:cc8191556294503c59c9bf373363f20ca593395c590f623e46412b9b4446867d","observation_id":"21158904-90a0-4cd4-874d-a9b7fab6cf84","resolution":{"observed_at":"2026-08-01T06:17:30.720072Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2009.10297","last_updated":"2020-09-27T04:07:11Z","snapshot_observed_at":"2026-08-12T14:26:33.940158Z","submitted_at":"2020-09-22T03:10:49Z","title":"CodeBLEU: a Method for Automatic Evaluation of Code Synthesis","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.10297","snapshot_observed_at":"2026-08-01T06:17:29.063792Z","title":null,"venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2607.21971","last_updated":"2026-07-24T04:35:29Z","snapshot_observed_at":"2026-08-15T08:18:09.144189Z","submitted_at":"2026-07-24T04:35:29Z","title":"Teaching LLMs to Self-Evolve: Cultivating Core Meta-Skills with Reinforcement Learning","version":1},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-01T06:17:29.063792Z"},"links":{"cited_paper":"/paper/2009.10297","citing_paper":"/paper/2607.21971"},"observation_digest":"sha256:cf0f72e121eb0353e5977b331a332199ab7746a35dc3f83ab40628d16e41e339","observation_id":"7628df11-b20d-406b-bd58-a5252f404447","resolution":{"observed_at":"2026-08-01T06:17:29.063792Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T06:17:28.376050Z","title":"Ale-bench: A benchmark for long-horizon objective-driven algorithm engineering","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21971","last_updated":"2026-07-24T04:35:29Z","snapshot_observed_at":"2026-08-15T08:18:09.144189Z","submitted_at":"2026-07-24T04:35:29Z","title":"Teaching LLMs to Self-Evolve: Cultivating Core Meta-Skills with Reinforcement Learning","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-01T06:17:28.376050Z"},"links":{"citing_paper":"/paper/2607.21971"},"observation_digest":"sha256:71fab0e178fdd0950c535ad85b2be45ea773efd1145d6e331293ff97729a7fab","observation_id":"ceee4493-3caf-4ecd-bbb1-9534e9b2cfe0","resolution":{"observed_at":"2026-08-01T06:17:28.376050Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2603.00729","last_updated":"2026-02-28T16:25:04Z","snapshot_observed_at":"2026-08-14T11:50:51.747505Z","submitted_at":"2026-02-28T16:25:04Z","title":"Qwen3-Coder-Next Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2603.00729","snapshot_observed_at":"2026-08-01T06:17:28.013414Z","title":"Qwen3-coder-next technical report.arXiv preprint arXiv:2603.00729,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21971","last_updated":"2026-07-24T04:35:29Z","snapshot_observed_at":"2026-08-15T08:18:09.144189Z","submitted_at":"2026-07-24T04:35:29Z","title":"Teaching LLMs to Self-Evolve: Cultivating Core Meta-Skills with Reinforcement Learning","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-01T06:17:28.013414Z"},"links":{"cited_paper":"/paper/2603.00729","citing_paper":"/paper/2607.21971"},"observation_digest":"sha256:b2c2e66277a9dff2661856ae792008b87a86a1a934a43fa17929f363b1e27c81","observation_id":"bf6c3b84-5ab0-4514-9465-41eb81c205c5","resolution":{"observed_at":"2026-08-01T06:17:28.013414Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T06:17:28.551317Z","title":"Deltaevolve: Accelerating scientific discovery through momentum-driven evolution.arXiv preprint arXiv:2602.02919,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21971","last_updated":"2026-07-24T04:35:29Z","snapshot_observed_at":"2026-08-15T08:18:09.144189Z","submitted_at":"2026-07-24T04:35:29Z","title":"Teaching LLMs to Self-Evolve: Cultivating Core Meta-Skills with Reinforcement Learning","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-01T06:17:28.551317Z"},"links":{"citing_paper":"/paper/2607.21971"},"observation_digest":"sha256:90d7131266ef2bab3d3dd3c959ea57c5a759f167a9112aaef003e26a5c8c82aa","observation_id":"4a99526a-bf74-4095-8a51-4b4eb8ceb6b6","resolution":{"observed_at":"2026-08-01T06:17:28.551317Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.16720","last_updated":"2026-04-30T02:46:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-21T18:04:31Z","title":"OpenAI o1 System Card","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.16720","snapshot_observed_at":"2026-08-01T06:17:28.866319Z","title":"Long Ouyang, Jeffrey Wu, Xu Jiang, Diogo Almeida, Carroll Wainwright, Pamela Mishkin, Chong Zhang, Sandhini Agarwal, Katarina Slama, Alex Ray, et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21971","last_updated":"2026-07-24T04:35:29Z","snapshot_observed_at":"2026-08-15T08:18:09.144189Z","submitted_at":"2026-07-24T04:35:29Z","title":"Teaching LLMs to Self-Evolve: Cultivating Core Meta-Skills with Reinforcement Learning","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-01T06:17:28.866319Z"},"links":{"cited_paper":"/paper/2412.16720","citing_paper":"/paper/2607.21971"},"observation_digest":"sha256:e5d6cd956cfbad7f34ea4381cdbcc7223dba177ea75fd98f59f39159b0266744","observation_id":"c328d9e8-e157-4951-b19d-0384681ffe9e","resolution":{"observed_at":"2026-08-01T06:17:28.866319Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01307","last_updated":"2025-08-15T15:21:46Z","snapshot_observed_at":"2026-08-14T13:46:28.086398Z","submitted_at":"2025-03-03T08:46:22Z","title":"Cognitive Behaviors that Enable Self-Improving Reasoners, or, Four Habits of Highly Effective STaRs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01307","snapshot_observed_at":"2026-08-01T06:17:28.178616Z","title":"Cognitive behaviors that enable self-improving reasoners, or, four habits of highly effective stars.arXiv preprint arXiv:2503.01307,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21971","last_updated":"2026-07-24T04:35:29Z","snapshot_observed_at":"2026-08-15T08:18:09.144189Z","submitted_at":"2026-07-24T04:35:29Z","title":"Teaching LLMs to Self-Evolve: Cultivating Core Meta-Skills with Reinforcement Learning","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-01T06:17:28.178616Z"},"links":{"cited_paper":"/paper/2503.01307","citing_paper":"/paper/2607.21971"},"observation_digest":"sha256:d765a90b17f83ed3d773c72f88818d9abef43ba2ff7f2b2f9c97850db9c95b13","observation_id":"4276ef4b-d744-4847-8398-4e57ee03d566","resolution":{"observed_at":"2026-08-01T06:17:28.178616Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.08073","last_updated":"2022-12-15T06:19:23Z","snapshot_observed_at":"2026-08-16T03:49:00.703994Z","submitted_at":"2022-12-15T06:19:23Z","title":"Constitutional AI: Harmlessness from AI Feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.08073","snapshot_observed_at":"2026-08-01T06:17:27.935566Z","title":"Constitu- tional AI: Harmlessness from AI feedback.arXiv preprint arXiv:2212.08073,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21971","last_updated":"2026-07-24T04:35:29Z","snapshot_observed_at":"2026-08-15T08:18:09.144189Z","submitted_at":"2026-07-24T04:35:29Z","title":"Teaching LLMs to Self-Evolve: Cultivating Core Meta-Skills with Reinforcement Learning","version":1},"reference_index":2026,"source":"pdf_text","source_observed_at":"2026-08-01T06:17:27.935566Z"},"links":{"cited_paper":"/paper/2212.08073","citing_paper":"/paper/2607.21971"},"observation_digest":"sha256:672e814d04128b5c94a98fe0b2d5594404504dfdde1c8b9ee680cc5b45023c9e","observation_id":"de564314-a21e-4d84-82d8-c59ab98209fc","resolution":{"observed_at":"2026-08-01T06:17:27.935566Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2607.21971","last_updated":"2026-07-24T04:35:29Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-15T08:18:09.144189Z","submitted_at":"2026-07-24T04:35:29Z","title":"Teaching LLMs to Self-Evolve: Cultivating Core Meta-Skills with Reinforcement Learning"},"reference_resolution":{"displayed":31,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":31,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":31},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"thesis":"As of 16 August 2026, this Paper Citation Record lists 31 of 31 outbound references and 0 inbound Pith citation observations for arXiv:2607.21971."}