{"as_of":"2026-08-11T21:11:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:c47406ddba1d5488bae27fd0702af75ccb537e72f39dc579a7d413f5a0f94b06","coverage":[{"denominator":89,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":89,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-23T02:41:21.571824Z","state":"measured"},{"denominator":91,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":91,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T11:35:11.631063Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-06T11:35:19.254627Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"cited_work":{"arxiv_id":"2502.12272","doi":null,"metadata_source":"pith","pith_arxiv_id":"2502.12272","snapshot_observed_at":"2026-08-06T11:35:19.254627Z","title":"Learning to Reason at the Frontier of Learnability","venue":"cs.LG","work_id":"dc38f080-bf90-4be4-994a-9744ff5a3437","year":2025},"citing_paper":{"arxiv_id":"2507.22607","last_updated":"2025-07-31T09:09:45Z","snapshot_observed_at":"2026-08-09T20:47:50.407542Z","submitted_at":"2025-07-30T12:23:21Z","title":"VL-Cogito: Progressive Curriculum Reinforcement Learning for Advanced Multimodal Reasoning","version":2},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-06T11:35:11.631063Z"},"links":{"cited_paper":"/paper/2502.12272","citing_paper":"/paper/2507.22607"},"observation_digest":"sha256:87269b7f60031ba9736f858b56b57fadc3ca4149e104e91a2341f68d70c31e29","observation_id":"12e36423-71e4-45c6-ba43-8731c8fc571f","resolution":{"observed_at":"2026-08-06T11:35:19.260704Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.12272","snapshot_observed_at":"2026-08-03T04:17:27.154486Z","title":"Learning to reason at the frontier of learnability.arXiv preprint arXiv:2502.12272, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.05547","last_updated":"2026-08-05T16:57:37Z","snapshot_observed_at":"2026-08-08T23:12:50.234380Z","submitted_at":"2026-02-05T11:06:37Z","title":"Multi-Task GRPO: Reliable LLM Reasoning Across Tasks","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-03T04:17:27.154486Z"},"links":{"cited_paper":"/paper/2502.12272","citing_paper":"/paper/2602.05547"},"observation_digest":"sha256:20ee7c0d87ff58d885314a0d92024281167e6cb51e36210b0d6fa48210e26a03","observation_id":"2e123e76-4eb5-44c5-8243-1d08c19a53f0","resolution":{"observed_at":"2026-08-03T04:17:27.154486Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2502.12272/citation-record","integrity":"/paper/2502.12272/integrity","json":"/paper/2502.12272/citation-record.json","paper":"/paper/2502.12272"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2501.12948","doi":"10.1016/j.artmed.2024.103001","metadata_source":"pith","pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","venue":"cs.CL","work_id":"e6b75ad5-2877-4168-97c8-710407094d20","year":2025},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:26a750b5374bb314f0a6d4a2c88ec54e951fc83cb254f1a8688195889d55d752","observation_id":"4f7f7a4c-d5f1-4184-8a17-8f69d175f469","resolution":{"observed_at":"2026-05-23T02:42:26.069190Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15124","last_updated":"2025-04-14T22:39:09Z","snapshot_observed_at":"2026-08-10T16:05:13.426341Z","submitted_at":"2024-11-22T18:44:04Z","title":"Tulu 3: Pushing Frontiers in Open Language Model Post-Training","version":5},"cited_work":{"arxiv_id":"2411.15124","doi":"10.48550/arxiv.2411.15124","metadata_source":"pith","pith_arxiv_id":"2411.15124","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Tulu 3: Pushing Frontiers in Open Language Model Post-Training","venue":"cs.CL","work_id":"28c9dbea-056a-48c2-8000-85f809827e45","year":2024},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2411.15124","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:89edb224011ffdd1ee295b48b32aedc0758d2d9e601bcc6c2623065a240d1731","observation_id":"c2cbddbf-206a-4b79-9e27-6848eb81e78e","resolution":{"observed_at":"2026-05-23T02:42:26.038058Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-23T21:53:00.522112+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T21:53:00.522112+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Learning to reason with llms","venue":null,"work_id":"bae5581f-90bc-49ab-b0f8-43401921dfca","year":2024},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:5d78fb55f37997937a3920ce8ec1e323e1b060686c698f2ff662526d9ea985c9","observation_id":"fdaf63a6-feec-45ff-9f96-01aa0bad26f3","resolution":{"observed_at":"2026-05-23T02:47:27.556981Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.20783","last_updated":"2025-10-06T09:30:03Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-26T17:59:14Z","title":"Understanding R1-Zero-Like Training: A Critical Perspective","version":2},"cited_work":{"arxiv_id":"2503.20783","doi":"10.48550/arxiv.2503.20783","metadata_source":"pith","pith_arxiv_id":"2503.20783","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding R1-Zero-Like Training: A Critical Perspective","venue":"cs.LG","work_id":"ec354f3b-9484-4a0c-94c8-92d4d0260835","year":2025},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2503.20783","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:e049a4fa5901d9a97061f252ddaf70f96ca0a88886fc474172b17a461d4e7b93","observation_id":"b290130e-c07f-4672-9a7d-35b03ff5ed0b","resolution":{"observed_at":"2026-05-23T02:42:26.030208Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-24T09:23:05.84445+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-24T09:23:05.84445+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Vineppo: Unlocking rl potential for llm reasoning through refined credit assignment","venue":null,"work_id":"01c63620-3179-441e-a7bd-a39be4b564b8","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:a7af7e6aaa780464ee45d10a21a05ebae4681fe2a1c607cb7e80eed37c0a1602","observation_id":"7046d60f-b54a-4c5a-a42b-cca48dcecaad","resolution":{"observed_at":"2026-05-23T02:47:27.553779Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.01679","last_updated":"2025-06-03T20:51:06Z","snapshot_observed_at":"2026-08-05T20:06:13.107818Z","submitted_at":"2024-10-02T15:49:30Z","title":"VinePPO: Refining Credit Assignment in RL Training of LLMs","version":2},"cited_work":{"arxiv_id":"2410.01679","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.01679","snapshot_observed_at":"2026-07-09T11:16:11.296273Z","title":"Available: https://arxiv.org/abs/2410.01679","venue":"cs.LG","work_id":"1b370c79-344d-4e23-8065-f313dbdc84de","year":2024},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2410.01679","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:8ff73c905b7e65ec80b5cf65d8293143de686f115f55e2a75e4c86225136bda8","observation_id":"66fa54c1-421c-4938-8538-0c19b137cf03","resolution":{"observed_at":"2026-05-23T02:42:26.025966Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20304","last_updated":"2024-05-30T17:50:04Z","snapshot_observed_at":"2026-08-10T22:18:58.746397Z","submitted_at":"2024-05-30T17:50:04Z","title":"Group Robust Preference Optimization in Reward-free RLHF","version":1},"cited_work":{"arxiv_id":"2405.20304","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2405.20304","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Group robust preference optimization in reward-free rlhf","venue":null,"work_id":"c3ea3ccd-d175-4914-95b0-9b717591d6af","year":2024},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2405.20304","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:7b18e1b233607e75ddd3633be0dd070ce53f5869dea7e37e744d8c4f12dd3b2a","observation_id":"79f22151-c393-4174-970b-f15eaa667fd7","resolution":{"observed_at":"2026-05-23T02:42:26.236094Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":"1707.06347","doi":"10.1016/j.artint.2010.12.005","metadata_source":"pith","pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Proximal Policy Optimization Algorithms","venue":"cs.LG","work_id":"240c67fe-d14d-4520-91c1-38a4e272ca19","year":2017},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:b313d755c07887f91da1ffe11771351060abf73f73d36fd666e0f5d7e939abb6","observation_id":"b2f396c2-77ad-4b05-8ab4-ceb0fcc0b99c","resolution":{"observed_at":"2026-05-23T02:42:26.231736Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-07T01:45:38.840969Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":"2110.14168","doi":"10.1002/j.1545-","metadata_source":"pith","pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Training Verifiers to Solve Math Word Problems","venue":"cs.LG","work_id":"acab1aa8-b4d6-40e0-a3ee-25341701dca2","year":2021},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:2da36bc4b92a0a96ddf8adf1e67e930f6892a6e4b5bab8d5f53e689850af0197","observation_id":"137dd64d-fe7f-46dd-bbdb-6ecd5ba5c84e","resolution":{"observed_at":"2026-05-23T02:42:26.227411Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.03874","last_updated":"2021-11-08T21:30:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-03-05T18:59:39Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","version":2},"cited_work":{"arxiv_id":"2103.03874","doi":"10.48550/arxiv.2103.03874","metadata_source":"pith","pith_arxiv_id":"2103.03874","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","venue":"cs.LG","work_id":"50652ac6-fb7c-4675-a2c2-159c241feb17","year":2021},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2103.03874","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:c20a44ae37f24d98928118194a07d045131da505a4d52e216ab70f27bd7fd26f","observation_id":"42f79281-0270-42b9-9207-899617751327","resolution":{"observed_at":"2026-05-23T02:42:26.081240Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:22.649941+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:22.649941+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24290","last_updated":"2025-07-05T09:01:04Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-31T16:36:05Z","title":"Open-Reasoner-Zero: An Open Source Approach to Scaling Up Reinforcement Learning on the Base Model","version":2},"cited_work":{"arxiv_id":"2503.24290","doi":"10.48550/arxiv.2503.24290","metadata_source":"pith","pith_arxiv_id":"2503.24290","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open-Reasoner-Zero: An Open Source Approach to Scaling Up Reinforcement Learning on the Base Model","venue":"cs.LG","work_id":"763e0e44-40dd-4bdd-8414-21f8f9ce6d10","year":2025},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2503.24290","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:7555406f84a5cd303fbbc43dcbf7c653be896d304173e8ddfd9b864495f3f561","observation_id":"0c878d87-271b-49f9-a748-107161b0410d","resolution":{"observed_at":"2026-05-23T02:42:26.222795Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.07965","last_updated":"2025-01-08T09:07:54Z","snapshot_observed_at":"2026-08-10T07:06:40.041118Z","submitted_at":"2024-04-11T17:52:01Z","title":"Rho-1: Not All Tokens Are What You Need","version":4},"cited_work":{"arxiv_id":"2404.07965","doi":"10.48550/arxiv.2404.07965","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.07965","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Rho-1: Not all tokens are what you need","venue":"arXiv (Cornell University)","work_id":"74e3da93-f1b3-41df-bc52-f8092041fb58","year":2025},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2404.07965","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:2785c57538516fe01401ec8eb354a06760690e86e4c02d61ec2c4072e25b2261","observation_id":"95ee720f-f476-4f9e-b569-0f3e85e7c30c","resolution":{"observed_at":"2026-05-23T02:42:26.218182Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Qwen2.5 technical report","venue":null,"work_id":"71f8c882-c442-4947-ae66-b3442d2519b1","year":2025},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:c966da9253157b9c5771ad9b859bc50f7fb94fdd733d5bc055763888770aac7f","observation_id":"f955939b-2b2a-4736-afb8-5f09842eb4d9","resolution":{"observed_at":"2026-05-23T02:47:27.549825Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.02884","last_updated":"2024-03-05T11:42:59Z","snapshot_observed_at":"2026-08-09T12:20:46.679035Z","submitted_at":"2024-03-05T11:42:59Z","title":"MathScale: Scaling Instruction Tuning for Mathematical Reasoning","version":1},"cited_work":{"arxiv_id":"2403.02884","doi":"10.48550/arxiv.2403.02884","metadata_source":"arxiv_reference","pith_arxiv_id":"2403.02884","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mathscale: Scaling instruction tuning for mathematical reasoning","venue":"arXiv (Cornell University)","work_id":"450d5342-49ed-44d9-85e1-41826683418c","year":2024},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2403.02884","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:62d37c9a266ff2624d754abd59d2de88b41c0877a9105fce1bf5aaf5886258da","observation_id":"6a7030aa-b42a-4730-9853-349c3df3b941","resolution":{"observed_at":"2026-05-23T02:42:26.213553Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14008","last_updated":"2024-06-06T13:19:44Z","snapshot_observed_at":"2026-08-03T03:39:09.398343Z","submitted_at":"2024-02-21T18:49:26Z","title":"OlympiadBench: A Challenging Benchmark for Promoting AGI with Olympiad-Level Bilingual Multimodal Scientific Problems","version":2},"cited_work":{"arxiv_id":"2402.14008","doi":"10.48550/arxiv.2402.14008","metadata_source":"pith","pith_arxiv_id":"2402.14008","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OlympiadBench: A Challenging Benchmark for Promoting AGI with Olympiad-Level Bilingual Multimodal Scientific Problems","venue":"cs.CL","work_id":"19abed3b-0ff6-409b-aded-a50205319aa3","year":2024},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2402.14008","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:2b06fc233b4180751c202e5ac7dbf379644e9f5d1d549629cb4873a4d52d2c68","observation_id":"8393f775-1f3a-43ca-937d-3c5d42a98506","resolution":{"observed_at":"2026-05-23T02:42:26.208213Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14740","last_updated":"2024-02-26T18:26:25Z","snapshot_observed_at":"2026-08-09T14:30:33.899591Z","submitted_at":"2024-02-22T17:52:34Z","title":"Back to Basics: Revisiting REINFORCE Style Optimization for Learning from Human Feedback in LLMs","version":2},"cited_work":{"arxiv_id":"2402.14740","doi":null,"metadata_source":"pith","pith_arxiv_id":"2402.14740","snapshot_observed_at":"2026-07-09T08:56:06.435604Z","title":"Back to Basics: Revisiting REINFORCE Style Optimization for Learning from Human Feedback in LLMs","venue":"cs.LG","work_id":"7bb8f9ec-1241-4472-a4fa-c636c6d79892","year":2024},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2402.14740","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:0d3c02ae11f35db65b095f4006067a181915c466ed26377543bbcde3bf1009a3","observation_id":"e66f894e-8ce2-45d8-b1a6-37b0f8284d24","resolution":{"observed_at":"2026-05-23T02:42:26.045213Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.12877","last_updated":"2023-04-25T14:49:34Z","snapshot_observed_at":"2026-08-08T18:35:34.296107Z","submitted_at":"2023-04-25T14:49:34Z","title":"Proximal Curriculum for Reinforcement Learning Agents","version":1},"cited_work":{"arxiv_id":"2304.12877","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.12877","snapshot_observed_at":"2026-07-04T18:30:01.557405Z","title":"Proximal curriculum for reinforcement learning agents","venue":null,"work_id":"3ce922ea-609e-413a-8854-eaf2fe8de7ab","year":2023},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2304.12877","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:c4a6b623b3adfe8944d32f85dbf3f4e94a580b3e9ffb0420156a73672a54bbc0","observation_id":"f494981d-ec13-4c91-babe-835993ab72ce","resolution":{"observed_at":"2026-05-23T02:42:26.073171Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Automatic goal generation for reinforcement learning agents","venue":null,"work_id":"22b828a3-6d38-4cc8-acd3-39dd737d21b2","year":2018},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:ad86719dd3b2e850c03fd3cc8379fa19e5c5f78e254219b0c45232eca7a8e931","observation_id":"ebdc1395-d93e-4c1b-9d96-bd4315a448ef","resolution":{"observed_at":"2026-05-23T02:47:27.523015Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15099","last_updated":"2024-10-29T18:25:44Z","snapshot_observed_at":"2026-08-05T16:37:23.069533Z","submitted_at":"2024-08-27T14:31:54Z","title":"No Regrets: Investigating and Improving Regret Approximations for Curriculum Discovery","version":3},"cited_work":{"arxiv_id":"2408.15099","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15099","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"No regrets: Investigating and improving regret approximations for curriculum discovery","venue":null,"work_id":"52cfcb90-5946-49d9-9b3a-531c79b0a64b","year":2024},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2408.15099","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:211a654b7e0fe7a9f71b06735f7620b817e95cc8bd5691a3925d122c62abfa28","observation_id":"1dce26a7-21f9-435d-a4e5-30ef10cb0337","resolution":{"observed_at":"2026-05-23T02:42:26.141497Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1007/bf00992696","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:47:34.373100Z","title":"Williams","venue":"Machine Learning","work_id":"469b3b81-55f9-4542-9dd7-570a63cfda74","year":1992},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:b6d5c0ee491fa724d407b6ba443ad03adf41ddb398d80601752a7b80d0c87ce5","observation_id":"6e35180a-fdf5-4874-95a9-d1a9af850592","resolution":{"observed_at":"2026-05-23T02:42:25.514626Z","resolver_source":"doi","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-07T21:08:27.931327+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-07T21:08:27.931327+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.18252","last_updated":"2025-04-26T08:33:32Z","snapshot_observed_at":"2026-08-07T20:55:15.330079Z","submitted_at":"2024-10-23T19:59:50Z","title":"Asynchronous RLHF: Faster and More Efficient Off-Policy RL for Language Models","version":3},"cited_work":{"arxiv_id":"2410.18252","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.18252","snapshot_observed_at":"2026-07-10T21:57:39.335341Z","title":"Asynchronous rlhf: Faster and more efficient off-policy rl for language models","venue":"cs.LG","work_id":"89e577b6-ccb0-422f-9751-3ad83561b117","year":2024},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2410.18252","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:23a3de8d32346f77a90b640e75724e7b9737b055f72ccf6840961c1588b0ac59","observation_id":"dca9b697-bd7c-4ec7-9771-f0cff09a2db2","resolution":{"observed_at":"2026-05-23T02:42:26.179007Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.11143","last_updated":"2025-10-09T12:22:46Z","snapshot_observed_at":"2026-07-31T12:28:37.704994Z","submitted_at":"2024-05-20T01:04:40Z","title":"OpenRLHF: An Easy-to-use, Scalable and High-performance RLHF Framework","version":6},"cited_work":{"arxiv_id":"2405.11143","doi":"10.48550/arxiv.2405.11143","metadata_source":"pith","pith_arxiv_id":"2405.11143","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OpenRLHF: An Easy-to-use, Scalable and High-performance RLHF Framework","venue":"cs.AI","work_id":"70fa48c9-2f84-49f6-9aca-37476e021fc3","year":2024},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2405.11143","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:f693d48fe51d05562ae18a34186497fee81379b1c9703a303c05d6407deebc68","observation_id":"aaeb81a3-ee00-4047-b28f-a9f6e50632e0","resolution":{"observed_at":"2026-05-23T02:42:26.089482Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"There may not be aha moment in r1-zero-like training — a pilot study","venue":null,"work_id":"08ad3897-16ac-4982-bc95-5b9709d32f56","year":2025},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:d05746e68a3cc613a3c8e658735fc1c182f765f05d3c6f47ba12c62c2c9258b4","observation_id":"10e0113a-ee17-49aa-9fcc-e8b7130c4ad7","resolution":{"observed_at":"2026-05-23T02:47:27.519337Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Numinamath","venue":null,"work_id":"56dbd789-1e68-4f24-852a-4b84cf032a17","year":2024},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:a7b2b60c2bf956b6c687cab3e224403d997bb9e1be498d851d8c714e2a8d029b","observation_id":"ee5c9610-11d4-4c2e-856c-651230236188","resolution":{"observed_at":"2026-05-23T02:47:27.573705Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.14858","last_updated":"2022-07-01T02:15:12Z","snapshot_observed_at":"2026-08-05T15:41:22.691461Z","submitted_at":"2022-06-29T18:54:49Z","title":"Solving Quantitative Reasoning Problems with Language Models","version":2},"cited_work":{"arxiv_id":"2206.14858","doi":"10.48550/arxiv.2206.14858","metadata_source":"pith","pith_arxiv_id":"2206.14858","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Solving Quantitative Reasoning Problems with Language Models","venue":"cs.CL","work_id":"17214d12-1ca8-4186-806d-53c6715383a0","year":2022},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2206.14858","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:6089794ffb6cd6d12acc23ecd342195aa24dc6dd018d02ffa93d5465110e7d9b","observation_id":"d8c04222-da3e-444d-b7b9-697610f088e4","resolution":{"observed_at":"2026-05-23T02:42:26.053039Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:08.21276+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:08.21276+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04642","last_updated":"2024-03-07T16:36:29Z","snapshot_observed_at":"2026-08-10T12:58:47.242770Z","submitted_at":"2024-03-07T16:36:29Z","title":"Teaching Large Language Models to Reason with Reinforcement Learning","version":1},"cited_work":{"arxiv_id":"2403.04642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04642","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Teaching large language models to reason with reinforcement learning","venue":null,"work_id":"ab9d8347-574c-4fbb-b8e1-44eeba9c66b9","year":2024},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2403.04642","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:ed87f1c52f29024ab2c5dcd7dcaa2d0c4ff323f63374e0a49c1344f9248af48d","observation_id":"23390e2f-76d4-4c61-b3fb-2f047a4552c5","resolution":{"observed_at":"2026-05-23T02:42:26.085841Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.03934","last_updated":"2021-06-12T10:50:10Z","snapshot_observed_at":"2026-07-06T10:02:35.988410Z","submitted_at":"2020-10-08T12:46:57Z","title":"Prioritized Level Replay","version":4},"cited_work":{"arxiv_id":"2010.03934","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2010.03934","snapshot_observed_at":"2026-07-04T14:09:52.695084Z","title":"Prioritized level replay","venue":null,"work_id":"791deccd-53da-4462-994c-5b5ae795d4ae","year":2021},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2010.03934","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:9ac9827ccb8531f6bfc503b559e405c2c19f13f145408394cfe9944224c37488","observation_id":"e13af358-449a-4dc1-b110-c6013ee8c5fb","resolution":{"observed_at":"2026-05-23T02:42:26.065432Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1812.03381","last_updated":"2018-12-08T20:16:16Z","snapshot_observed_at":"2026-07-06T07:19:59.972802Z","submitted_at":"2018-12-08T20:16:16Z","title":"Learning Montezuma's Revenge from a Single Demonstration","version":1},"cited_work":{"arxiv_id":"1812.03381","doi":"10.48550/arxiv.1812.03381","metadata_source":"pith","pith_arxiv_id":"1812.03381","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Learning Montezuma's Revenge from a Single Demonstration","venue":"cs.LG","work_id":"678d4efb-de3e-44d4-b506-208158d68e08","year":2018},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/1812.03381","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:2f02067658be3a51e29aafba3f4e2be169a18760b68295c501027369328a876e","observation_id":"d59497fd-ba5c-400e-83c3-3288b371159f","resolution":{"observed_at":"2026-05-23T02:42:26.057246Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.13228","last_updated":"2024-07-03T13:46:33Z","snapshot_observed_at":"2026-08-08T16:03:10.053914Z","submitted_at":"2024-02-20T18:42:34Z","title":"Smaug: Fixing Failure Modes of Preference Optimisation with DPO-Positive","version":2},"cited_work":{"arxiv_id":"2402.13228","doi":null,"metadata_source":"pith","pith_arxiv_id":"2402.13228","snapshot_observed_at":"2026-07-03T16:48:40.350225Z","title":"Smaug: Fixing Failure Modes of Preference Optimisation with DPO-Positive","venue":"cs.CL","work_id":"b220de8b-d5cf-4eef-9841-1428c753012c","year":2024},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2402.13228","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:4a34d6caa2f69a2e254a1d124c59420df48859955b2377de5264a772e136d52a","observation_id":"b9767396-8989-4e33-86e3-0366a0ea8681","resolution":{"observed_at":"2026-05-23T02:42:26.061849Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.14476","last_updated":"2025-05-20T01:37:34Z","snapshot_observed_at":"2026-08-02T01:40:54.187278Z","submitted_at":"2025-03-18T17:49:06Z","title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","version":2},"cited_work":{"arxiv_id":"2503.14476","doi":"10.48550/arxiv.2503.14476","metadata_source":"pith","pith_arxiv_id":"2503.14476","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","venue":"cs.LG","work_id":"64019d00-0b11-4bbd-b173-b46c8fad0157","year":2025},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2503.14476","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:ef50a6829f7d4ffacc1a271f0db64c56ba7fc2ef9c51a2c4f2bba2230be68c45","observation_id":"5ba38af3-159d-467c-9c25-3b7f99c5bbf9","resolution":{"observed_at":"2026-05-23T02:42:26.203304Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-08T16:08:20.547492+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T16:08:20.547492+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12599","last_updated":"2025-06-03T02:14:54Z","snapshot_observed_at":"2026-08-11T01:31:26.242281Z","submitted_at":"2025-01-22T02:48:14Z","title":"Kimi k1.5: Scaling Reinforcement Learning with LLMs","version":4},"cited_work":{"arxiv_id":"2501.12599","doi":"10.48550/arxiv.2501.12599","metadata_source":"pith","pith_arxiv_id":"2501.12599","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Kimi k1.5: Scaling Reinforcement Learning with LLMs","venue":"cs.AI","work_id":"bff96ab1-bd6a-4585-be23-74fdb51969c7","year":2025},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2501.12599","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:dc641ee2b351f034b01864bab610f61cc8d0d63932c1ffc4e17fc648610bf6c8","observation_id":"a248f80a-a5e1-4951-93b9-13e1d36b7984","resolution":{"observed_at":"2026-05-23T02:42:26.188553Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-07-09T10:48:38.585868+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T10:48:38.585868+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.13818","last_updated":"2026-04-22T00:26:08Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-18T17:49:55Z","title":"Not All Rollouts are Useful: Down-Sampling Rollouts in LLM Reinforcement Learning","version":5},"cited_work":{"arxiv_id":"2504.13818","doi":"10.48550/arxiv.2504.13818","metadata_source":"pith","pith_arxiv_id":"2504.13818","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"Not All Rollouts are Useful: Down-Sampling Rollouts in LLM Reinforcement Learning","venue":"cs.LG","work_id":"e6d53e5b-2180-482b-82ca-0e64d572c87f","year":2025},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2504.13818","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:1d75c822042427ab58464c185bb7b6c5b22398109ed61573e1152ddde6655959","observation_id":"b7293fb5-3be6-4a7c-9258-bd6653cd1ae0","resolution":{"observed_at":"2026-05-23T02:42:26.183846Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The llama 4 herd: The beginning of a new era of natively multimodal ai innovation","venue":null,"work_id":"fdf44679-84b9-4316-85e9-ec5b4c2d27f5","year":2025},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:1242830ea8e16c7c8b27e42d447ea8f1f581c69e482855caa37d54865e692d1e","observation_id":"fccbdeb1-b453-4849-a485-80f6c0eee41e","resolution":{"observed_at":"2026-05-23T02:47:27.511125Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2012.02096","last_updated":"2021-02-04T03:01:31Z","snapshot_observed_at":"2026-08-11T03:11:26.110490Z","submitted_at":"2020-12-03T17:37:01Z","title":"Emergent Complexity and Zero-shot Transfer via Unsupervised Environment Design","version":2},"cited_work":{"arxiv_id":"2012.02096","doi":"10.48550/arxiv.2012.02096","metadata_source":"arxiv_reference","pith_arxiv_id":"2012.02096","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Emergent complexity and zero-shot transfer via unsupervised environment design","venue":"arXiv (Cornell University)","work_id":"13fa0948-1511-420f-912e-574ff5f55f77","year":2021},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2012.02096","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:8187e68cc79cee59653f5464ade27166e4b413241ae0d2ebe817a66d323c0dd5","observation_id":"017d8ccd-e4a3-4d4e-a1d6-29eda5c39b62","resolution":{"observed_at":"2026-05-23T02:42:26.198274Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-10T21:38:18.393196+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-10T21:38:18.393196+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Minigrid &amp; miniworld: Modular &amp; customizable reinforcement learning environments for goal-oriented tasks","venue":null,"work_id":"4c354294-1888-46f6-ab37-b86eec56e0ae","year":2023},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:7bb4f34e963f540dcc2a27fc9c6685d5080f0862d8e564f23556f3bcd64a633e","observation_id":"64e71307-ec78-4c6e-9ae0-e2a5656ffd7d","resolution":{"observed_at":"2026-05-23T02:47:27.567335Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.12044","last_updated":"2024-11-19T09:52:55Z","snapshot_observed_at":"2026-08-08T09:56:48.439696Z","submitted_at":"2023-12-19T10:57:12Z","title":"XLand-MiniGrid: Scalable Meta-Reinforcement Learning Environments in JAX","version":4},"cited_work":{"arxiv_id":"2312.12044","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.12044","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Xland-minigrid: Scalable meta-reinforcement learning environments in jax","venue":null,"work_id":"e0014edd-096c-4f70-b56f-63147a2459f0","year":2024},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2312.12044","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:9f58549cff34870dd7ce404f28cbb3794743a1b32bc518fb67854b32e54000da","observation_id":"2dbf633d-26d8-46a5-8cd3-57a13e2ed0ed","resolution":{"observed_at":"2026-05-23T02:42:26.048877Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.10090","last_updated":"2026-07-04T05:19:25Z","snapshot_observed_at":"2026-08-09T03:06:29.095333Z","submitted_at":"2023-11-16T18:58:43Z","title":"JaxMARL: Multi-Agent RL Environments and Algorithms in JAX","version":6},"cited_work":{"arxiv_id":"2311.10090","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.10090","snapshot_observed_at":"2026-07-07T02:17:02.673620Z","title":"Jaxmarl: Multi-agent rl environments and algorithms in jax","venue":null,"work_id":"2a3824b2-3a1e-4362-be0b-b58a12ec06d0","year":2024},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2311.10090","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:2333f2364dc2f358aa25d54cd0145af2139a464e7de8e135ad5c2ae3df8d07da","observation_id":"8d6e0c89-2840-4822-a6f3-7c0dbb8d85cb","resolution":{"observed_at":"2026-07-07T02:17:02.673620Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1606.01540","last_updated":"2016-06-05T17:54:48Z","snapshot_observed_at":"2026-08-09T23:24:21.399948Z","submitted_at":"2016-06-05T17:54:48Z","title":"OpenAI Gym","version":1},"cited_work":{"arxiv_id":"1606.01540","doi":"10.1109/jssc.2019","metadata_source":"pith","pith_arxiv_id":"1606.01540","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"OpenAI Gym","venue":"cs.LG","work_id":"6af98f3f-f074-41ae-a689-7dd7b4b8efde","year":2016},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/1606.01540","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:5d8ce10dd6f28febf420218d23b2d09b3d9f45b011431a663343a73ff0d06587","observation_id":"66e42759-9de4-40d8-8858-fdd92eea52f3","resolution":{"observed_at":"2026-05-23T02:42:26.173933Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"JAX: composable transformations of Python+NumPy programs","venue":null,"work_id":"a6623f60-5524-4152-889e-6618862f34e6","year":2018},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:36f70e40fbd203b97e6b9cccec0b6c41ad104cd20ef72beb8e9e9f3c4699dd4e","observation_id":"6a88b35b-2f4c-4967-ab08-229c79b51f0a","resolution":{"observed_at":"2026-05-23T02:47:27.507646Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.04368","last_updated":"2024-11-07T01:58:42Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-11-07T01:58:42Z","title":"Measuring short-form factuality in large language models","version":1},"cited_work":{"arxiv_id":"2411.04368","doi":"10.48550/arxiv.2411.04368","metadata_source":"pith","pith_arxiv_id":"2411.04368","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Measuring short-form factuality in large language models","venue":"cs.CL","work_id":"f8e490ab-7057-43fb-8c6d-06fc603836c7","year":2024},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2411.04368","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:2bb16f087f320f1876e2297e1f720bc9c15904f97d6516b3586d243558174c62","observation_id":"75b9796e-604a-43ed-a377-36da011eea82","resolution":{"observed_at":"2026-05-23T02:42:26.155841Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.01302","last_updated":"2023-09-30T18:36:42Z","snapshot_observed_at":"2026-07-06T12:43:24.521815Z","submitted_at":"2022-03-02T18:40:00Z","title":"Evolving Curricula with Regret-Based Environment Design","version":3},"cited_work":{"arxiv_id":"2203.01302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2203.01302","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Evolving curricula with regret-based environment design","venue":null,"work_id":"d7dfc56e-54c0-4912-a67d-aabdb2ad9a2d","year":2023},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2203.01302","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:8eb91976b4e293547b47686b6d0186e8bcb87ab4fc7a1b8b06a222567d69bcc8","observation_id":"b3d21010-bb44-459f-a500-7b07dff31324","resolution":{"observed_at":"2026-05-23T02:42:26.165070Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.03335","last_updated":"2025-10-16T08:23:36Z","snapshot_observed_at":"2026-08-11T20:12:00.210052Z","submitted_at":"2025-05-06T09:08:00Z","title":"Absolute Zero: Reinforced Self-play Reasoning with Zero Data","version":3},"cited_work":{"arxiv_id":"2505.03335","doi":"10.48550/arxiv.2505.03335","metadata_source":"pith","pith_arxiv_id":"2505.03335","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Absolute Zero: Reinforced Self-play Reasoning with Zero Data","venue":"cs.LG","work_id":"b59092c4-76ed-4c78-9006-312bde2e40a6","year":2025},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2505.03335","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:89c0ec089ef3f98fbca84050e54e255dd98aad82889660d8066ddf303bda87aa","observation_id":"e35f324e-bd90-4fd6-bc13-56172bca0ec5","resolution":{"observed_at":"2026-05-23T02:42:26.077279Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-23T21:52:59.097203+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T21:52:59.097203+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"9a5767f7-f5b6-4ac1-84e1-8d42bb7939d1","year":2018},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:e02166c1bc7ecae64dd90f7ce1f9279530a8e965df673af4dcd0923c2dfc8a27","observation_id":"ab6cbce9-9fc2-40f7-97e6-c4081d05ce6f","resolution":{"observed_at":"2026-05-23T02:47:27.563811Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Curriculum learning","venue":null,"work_id":"c5d50b44-2deb-46d3-9178-bba5a7b5d999","year":2009},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:83603ed111cc422343dd8097557aa87bc7254a547976f34a50b6248ab7f8566d","observation_id":"3de0a68b-6434-47f1-a568-07ee7e0a358c","resolution":{"observed_at":"2026-05-23T02:47:27.503948Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Learning and development in neural networks: The importance of starting small","venue":null,"work_id":"b42009bf-7d75-4bbc-8142-afd8777578c1","year":1993},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:7a42bc1982b33db45649f04875432ee9096fd5e51f356ac1e7af7cb27481cb54","observation_id":"e4019767-2d6d-40b9-84de-c43f166aca00","resolution":{"observed_at":"2026-05-23T02:47:27.500421Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Online batch selection for faster training of neural networks","venue":null,"work_id":"bdc8024c-826c-4e23-8abc-f7879656c8c4","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:6358cdadc3e0e31cebbe8233e9dc9e6f06aae9fe6f26c81577a4c9e490626573","observation_id":"0d215df8-c324-44a5-adaa-71f935de9a1d","resolution":{"observed_at":"2026-05-23T02:47:27.496724Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1511.06343","last_updated":"2016-04-25T14:00:21Z","snapshot_observed_at":"2026-08-10T13:42:16.073324Z","submitted_at":"2015-11-19T20:24:09Z","title":"Online Batch Selection for Faster Training of Neural Networks","version":4},"cited_work":{"arxiv_id":"1511.06343","doi":null,"metadata_source":"pith","pith_arxiv_id":"1511.06343","snapshot_observed_at":"2026-07-09T21:36:34.306361Z","title":"Online Batch Selection for Faster Training of Neural Networks","venue":"cs.LG","work_id":"35a8790d-539d-4362-b36d-6c27d43a6860","year":2015},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/1511.06343","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:065f0beab089fe2d0d4e198e5bbd2fb22d314407e18807b0b261d49424ef204e","observation_id":"bda077f8-05ae-47ac-951f-036e3f34c4e6","resolution":{"observed_at":"2026-05-23T02:42:26.116137Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1907.04371","last_updated":"2020-02-01T21:34:16Z","snapshot_observed_at":"2026-08-02T15:07:57.028674Z","submitted_at":"2019-07-09T19:09:51Z","title":"Ordered SGD: A New Stochastic Optimization Framework for Empirical Risk Minimization","version":5},"cited_work":{"arxiv_id":"1907.04371","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"1907.04371","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Ordered sgd: A new stochastic optimization framework for empirical risk minimization","venue":null,"work_id":"0a7faa3f-bb32-4da0-b8aa-37b089435795","year":2020},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/1907.04371","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:6a4fb5cc9b449ee480364ecc5d3f60bf6122b75a1ee050ca11ab167a69ebfb83","observation_id":"56f0f289-b948-4ea1-90a1-8512750e7c1d","resolution":{"observed_at":"2026-05-23T02:42:26.193403Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1910.00762","last_updated":"2019-10-02T03:34:29Z","snapshot_observed_at":"2026-08-07T07:17:09.592892Z","submitted_at":"2019-10-02T03:34:29Z","title":"Accelerating Deep Learning by Focusing on the Biggest Losers","version":1},"cited_work":{"arxiv_id":"1910.00762","doi":null,"metadata_source":"pith","pith_arxiv_id":"1910.00762","snapshot_observed_at":"2026-07-09T21:36:34.301049Z","title":"Accelerating deep learning by focusing on the biggest losers","venue":"cs.LG","work_id":"d74905ea-58c3-4017-8f2d-878a41346b29","year":2019},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/1910.00762","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:56b8505b4c4578316e097f6d19806cca566a2464b52fd74c7d9e74948aa8a81d","observation_id":"658cc2a6-ce97-4eb2-93cf-fe04e7e4dc14","resolution":{"observed_at":"2026-05-23T02:42:26.150995Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1802.03796","last_updated":"2018-06-08T18:04:50Z","snapshot_observed_at":"2026-07-06T06:22:47.916544Z","submitted_at":"2018-02-11T19:24:47Z","title":"Curriculum Learning by Transfer Learning: Theory and Experiments with Deep Networks","version":4},"cited_work":{"arxiv_id":"1802.03796","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"1802.03796","snapshot_observed_at":"2026-07-01T09:45:40.136469Z","title":"Curriculum learning by transfer learning: Theory and experiments with deep networks","venue":null,"work_id":"7626e1e3-853f-4769-b943-dc386d0da54a","year":2018},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/1802.03796","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:3a39998c6c631eabb867e45236c269a3d3d859d87d068e8a622e74e4315639ca","observation_id":"8390b0f2-a71e-42e1-9618-87e0c85d49cc","resolution":{"observed_at":"2026-05-23T02:42:26.169796Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Active learning literature survey","venue":null,"work_id":"afc51652-f571-4d67-ac8a-52526dc22b5a","year":2009},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:266c871640bdd2232cf7a14cb532c3e5bc51d51d26d1327010a42a4eef246f6b","observation_id":"adf83da6-8a75-4fd5-8bb9-4405c3e7615a","resolution":{"observed_at":"2026-05-23T02:47:27.492887Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Confidence-based active learning","venue":null,"work_id":"4ef1d214-f0d2-4f36-b6fc-99b9a2cfa557","year":2006},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:4b932e2c22ce4f4c21167e3c7378a77b8f878a77d0c59145d01d03377d232dc0","observation_id":"afc46c93-d264-4c35-9265-de39385a70cf","resolution":{"observed_at":"2026-05-23T02:47:27.489016Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1906.11829","last_updated":"2020-10-27T00:52:20Z","snapshot_observed_at":"2026-07-06T08:03:24.054624Z","submitted_at":"2019-06-26T23:01:47Z","title":"Selection via Proxy: Efficient Data Selection for Deep Learning","version":4},"cited_work":{"arxiv_id":"1906.11829","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"1906.11829","snapshot_observed_at":"2026-07-03T19:28:52.634253Z","title":"Selection via proxy: Efficient data se- lection for deep learning.arXiv preprint arXiv:1906.11829","venue":null,"work_id":"12daad0e-7402-46c0-bc01-a0fd7dfbadf7","year":1906},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/1906.11829","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:0fceea5da52a029a16cc9f23ded3e45ca878015df5490cc38cd8ec4e8c98c708","observation_id":"0cd69aea-9ee3-423a-9512-d3105fcdcb0a","resolution":{"observed_at":"2026-05-23T02:42:26.121575Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.07137","last_updated":"2022-09-26T17:28:16Z","snapshot_observed_at":"2026-07-06T13:20:51.362304Z","submitted_at":"2022-06-14T19:49:52Z","title":"Prioritized Training on Points that are Learnable, Worth Learning, and Not Yet Learnt","version":3},"cited_work":{"arxiv_id":"2206.07137","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2206.07137","snapshot_observed_at":"2026-06-30T22:05:05.665419Z","title":"Prioritized training on points that are learnable, worth learning, and not yet learnt","venue":null,"work_id":"d84eead5-4940-4998-8d33-0d0bdb183105","year":2022},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2206.07137","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:3a043a6d1721e11fcf6203167cfb4257e5a939d012fb2ebe225b02c4bfcbc416","observation_id":"b9113709-88e6-462b-9b33-c67eb55989bb","resolution":{"observed_at":"2026-05-23T02:42:26.126116Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1808.04759","last_updated":"2019-05-14T16:34:15Z","snapshot_observed_at":"2026-08-10T08:16:33.217364Z","submitted_at":"2018-08-14T15:45:48Z","title":"An Overview and a Benchmark of Active Learning for Outlier Detection with One-Class Classifiers","version":2},"cited_work":{"arxiv_id":"1808.04759","doi":null,"metadata_source":"pith","pith_arxiv_id":"1808.04759","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"An Overview and a Benchmark of Active Learning for Outlier Detection with One-Class Classifiers","venue":"cs.LG","work_id":"4c3800e8-6646-418b-9b46-2274f7cbd007","year":2018},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/1808.04759","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:42a491314365576dbae9aab458f9e9783b309e8dd90b7806980a2345671a85a4","observation_id":"898ef982-bd6f-4e4f-b954-3510565f7a36","resolution":{"observed_at":"2026-05-23T02:42:26.111402Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Training deep models faster with robust, approximate importance sampling","venue":null,"work_id":"421de673-3a9d-428d-9f1a-8675cee38ad3","year":2018},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:b634ede117e35001bc0130174e032bfb507d4c6327cd2653c8e180e00ce02d76","observation_id":"f064bdf0-fd28-4c58-9430-a0768496b8c0","resolution":{"observed_at":"2026-05-23T02:47:27.622517Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Not all samples are created equal: Deep learning with importance sampling","venue":null,"work_id":"77a9584e-eddb-4d2a-8996-f679cae25ec5","year":2018},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:25cddc616b1954a2a1f00e765ea6928f3c6f029fbd5d845eae1d5221298f6146","observation_id":"2a4d60cb-1c34-478e-bd5e-6b3a0e26c75d","resolution":{"observed_at":"2026-05-23T02:47:27.618558Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Self-paced learning for latent variable models","venue":null,"work_id":"a6496fab-4b6f-4710-98d4-274e7045732e","year":2010},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:fdf7620ea156e1f3e1d2621c33190d5f8174096bf9f12f8e7ce218375a9f54a4","observation_id":"341adf48-c1a3-4997-815f-3196b743385b","resolution":{"observed_at":"2026-05-23T02:47:27.634546Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1704.03003","last_updated":"2017-04-10T18:25:29Z","snapshot_observed_at":"2026-08-01T21:11:47.069345Z","submitted_at":"2017-04-10T18:25:29Z","title":"Automated Curriculum Learning for Neural Networks","version":1},"cited_work":{"arxiv_id":"1704.03003","doi":null,"metadata_source":"pith","pith_arxiv_id":"1704.03003","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Automated Curriculum Learning for Neural Networks","venue":"cs.NE","work_id":"3e4c1efa-45d7-4f02-8a5a-14a47714e09e","year":2017},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/1704.03003","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:e9efd309c28e30d9056b81f3b77be23a6ab97d3f20e37160db37250ffe7cc7bc","observation_id":"ad35adb3-e42d-4c0e-92c1-9c32b935e279","resolution":{"observed_at":"2026-05-23T02:42:26.130839Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1707.00183","last_updated":"2017-11-29T20:57:09Z","snapshot_observed_at":"2026-07-06T05:49:21.582343Z","submitted_at":"2017-07-01T18:13:17Z","title":"Teacher-Student Curriculum Learning","version":2},"cited_work":{"arxiv_id":"1707.00183","doi":null,"metadata_source":"pith","pith_arxiv_id":"1707.00183","snapshot_observed_at":"2026-06-28T23:52:49.216887Z","title":"Teacher-Student Curriculum Learning","venue":"cs.LG","work_id":"7c8c807e-3c64-4652-92a5-0bfcbb669352","year":2017},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/1707.00183","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:39dd82fd385d29573678e4c21340d91f61c3c705e3841d4f8d4e774f76c1a337","observation_id":"e480ceff-129f-4204-8390-55f1b6bf2d4f","resolution":{"observed_at":"2026-05-23T02:42:26.097838Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A survey of multi-task deep reinforcement learning","venue":null,"work_id":"3e21347e-ba41-414f-a783-c5d8e94e246b","year":2020},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:0a103a1c1ed7afbdf8f73370740d2e7dd6931cbdd272c0164ede5e5eb4761a9e","observation_id":"f7a01818-4359-47b6-a735-ded8010df04b","resolution":{"observed_at":"2026-05-23T02:47:27.611853Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Automatic curriculum learning through value disagreement","venue":null,"work_id":"a783c50f-5b1d-4e9a-a23f-9a327c73498d","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:74c9585b010107954bb233c78a5bb3b120e9f381fba7eeeb13774a87c05e5fe7","observation_id":"4d81e9e8-2b97-4458-813d-3fd6076f4007","resolution":{"observed_at":"2026-05-23T02:47:27.604553Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.09641","last_updated":"2020-06-17T03:58:25Z","snapshot_observed_at":"2026-07-06T09:29:54.899176Z","submitted_at":"2020-06-17T03:58:25Z","title":"Automatic Curriculum Learning through Value Disagreement","version":1},"cited_work":{"arxiv_id":"2006.09641","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2006.09641","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Available: https://arxiv.org/abs/2006.09641","venue":null,"work_id":"a15a584e-1aca-4b23-af6f-236059b7fb3a","year":2006},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2006.09641","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:23a1c2effed07f764ef93a76e22301dd3310ad1a1ec74cbf3b8688ce13aaac1d","observation_id":"c3e9b355-ff56-4a18-8a1b-b0961559ae68","resolution":{"observed_at":"2026-05-23T02:42:26.106419Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2011.01054","last_updated":"2021-07-01T13:45:56Z","snapshot_observed_at":"2026-08-05T12:52:28.696666Z","submitted_at":"2020-11-02T15:37:37Z","title":"Information-theoretic Task Selection for Meta-Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2011.01054","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2011.01054","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Information-theoretic task selection for meta-reinforcement learning","venue":null,"work_id":"f121ce98-6373-4a6f-b5aa-cdd744372e22","year":2021},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2011.01054","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:ce28a0e2fb70f1a38de9fb7da469811bbf1dcc4cb038ba12be7026fa0b9aef02","observation_id":"64a38f54-5c17-4084-a444-728db0e96793","resolution":{"observed_at":"2026-05-23T02:42:26.093984Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2007.02832","last_updated":"2020-07-06T15:36:05Z","snapshot_observed_at":"2026-08-03T21:30:52.105566Z","submitted_at":"2020-07-06T15:36:05Z","title":"Maximum Entropy Gain Exploration for Long Horizon Multi-goal Reinforcement Learning","version":1},"cited_work":{"arxiv_id":"2007.02832","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2007.02832","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Maximum entropy gain exploration for long horizon multi-goal reinforcement learning","venue":null,"work_id":"dd3f8939-373d-452c-b019-2651fe090a29","year":2020},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2007.02832","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:6108188808c244c18266a15d0842bfd966cf82fb66b9e33a63599d643aaf7dae","observation_id":"70dea3b6-cd80-46cf-8035-3d6f2862fc7a","resolution":{"observed_at":"2026-05-23T02:42:26.135156Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1903.03698","last_updated":"2020-08-04T04:07:27Z","snapshot_observed_at":"2026-07-06T07:38:07.496179Z","submitted_at":"2019-03-08T23:32:17Z","title":"Skew-Fit: State-Covering Self-Supervised Reinforcement Learning","version":4},"cited_work":{"arxiv_id":"1903.03698","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"1903.03698","snapshot_observed_at":"2026-07-04T06:59:38.343947Z","title":"arXiv preprint arXiv:1903.03698 , year=","venue":null,"work_id":"9912f351-bd16-4037-9627-98a975a611ff","year":1903},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/1903.03698","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:1f95b2833f28e0640c5878518dcb03629f022521f2e76a5b52edc83ff01d9c37","observation_id":"1f05f239-4449-48a3-b8f2-60e199228dce","resolution":{"observed_at":"2026-05-23T02:42:26.146328Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1901.09720","last_updated":"2019-03-25T10:04:05Z","snapshot_observed_at":"2026-07-06T07:29:28.340655Z","submitted_at":"2019-01-28T15:00:29Z","title":"CLIC: Curriculum Learning and Imitation for object Control in non-rewarding environments","version":4},"cited_work":{"arxiv_id":"1901.09720","doi":null,"metadata_source":"pith","pith_arxiv_id":"1901.09720","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"CLIC: Curriculum Learning and Imitation for object Control in non-rewarding environments","venue":"cs.LG","work_id":"492ecd38-47d2-4b49-8cee-c2b63a38e7e0","year":2019},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/1901.09720","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:ac2a6440739082641ab29ba3f1a0e5beab3e0f48c908bd53fb9981c9c3e90010","observation_id":"727e61de-24b1-4793-b80a-fdf8899f22fe","resolution":{"observed_at":"2026-05-23T02:42:26.101950Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.01114","last_updated":"2020-10-02T17:17:45Z","snapshot_observed_at":"2026-07-06T10:00:59.593985Z","submitted_at":"2020-10-02T17:17:45Z","title":"Goal-GAN: Multimodal Trajectory Prediction Based on Goal Position Estimation","version":1},"cited_work":{"arxiv_id":"2010.01114","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2010.01114","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Goal-gan: Multimodal trajectory prediction based on goal position estimation","venue":null,"work_id":"e8fe4b86-e2b9-48dd-9bf1-7294ea813fd4","year":2020},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2010.01114","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:55d6902c83f25d29f5769e5eda075497a66676fbdf4ff9f906aa116c8db3509f","observation_id":"c34e21d6-43e9-4e82-bc52-0fda7635b0e5","resolution":{"observed_at":"2026-05-23T02:42:26.034216Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1511.05952","last_updated":"2016-02-25T17:55:31Z","snapshot_observed_at":"2026-07-06T04:37:00.543211Z","submitted_at":"2015-11-18T20:54:44Z","title":"Prioritized Experience Replay","version":4},"cited_work":{"arxiv_id":"1511.05952","doi":"10.48550/arxiv.1511.05952","metadata_source":"pith","pith_arxiv_id":"1511.05952","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Prioritized Experience Replay","venue":"cs.LG","work_id":"927187c1-c50e-4ca7-b0fa-55589957731f","year":2015},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/1511.05952","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:ab232fbf2bb779c898a408b12e1ddf7ed87ca477d8f793f74d976db8b85d9856","observation_id":"7124d511-8153-4b2c-9e55-ffa1792fb33e","resolution":{"observed_at":"2026-05-23T02:42:26.041639Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In [48] the authors use the loss from a pre-trained model to estimate the difficulty of new samples for a freshly initialized network learning a new task","venue":null,"work_id":"3084f4f9-5cd6-44d3-bd25-274e2c33e000","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:36b5e31bd76c6fd62cc2e5287ddd2c531cf2268b24c31431f043d05863f044b0","observation_id":"cf3a376b-7f5e-449c-bb41-3270b1bba14a","resolution":{"observed_at":"2026-05-23T02:47:27.597336Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"LILO can be seen as using return variance—or learnability—as an estimator of entropy or uncertainty","venue":null,"work_id":"88f7773e-43c6-46f8-be64-42a932336ffc","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:e3d1fbc3a26f6beae252aeecae8fc657178c368441ac1f12396cae53b49ea9eb","observation_id":"e4942561-ca24-4a84-8d16-c660bc6881de","resolution":{"observed_at":"2026-05-23T02:47:27.580531Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"This allows prioritizing samples that maximize the change in loss—i.e., the model’s learning progress [ 52, 11, 53, 49]","venue":null,"work_id":"ddde022e-5e71-4662-900c-9654e5dffcb0","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:a57e3c87825a3a01d07ab418cac343dcdabdbb849b4956099acb11fe35cb36cd","observation_id":"e6b2d2c2-ed7d-4ac1-ad38-a25ab2db34a8","resolution":{"observed_at":"2026-05-23T02:47:27.585243Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Self-paced learning [56] is an early approach that allows the model to determine the pace at which it incorporates harder examples with higher values of U","venue":null,"work_id":"29f7bfda-d74b-484f-b502-d6f1afd3ac67","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:867a54fba00bb0fa1ec53b0fbce3ef04fa4e56b4f54a1c5f66d3c33cc031211d","observation_id":"249db8c9-8b4f-4918-ba51-3441497bc233","resolution":{"observed_at":"2026-05-23T02:47:27.608148Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Each bullet point contains a claim and a hyperlink to the section of the paper that proves the claim","venue":null,"work_id":"95be00f2-83cb-4e92-8c1b-962361088c8b","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:91dface5efe6f1a9f5953d503a9695343aa0d98e0a803b7079889c19b55b0bd6","observation_id":"c813b16f-bd7c-49aa-a48c-5d1e9b73ad32","resolution":{"observed_at":"2026-05-23T02:47:27.534779Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Section 7 also contains some limitations","venue":null,"work_id":"72919680-a787-4129-a892-d323813cc5dd","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:90ca7420469733722688d5556b15d48cac1ce1d2a2ad73dc403298832f2bbe96","observation_id":"5eb58b2e-1dfe-408b-a88e-b0a9712c4ceb","resolution":{"observed_at":"2026-05-23T02:47:27.527386Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Guidelines: • The answer NA means that the paper does not include theoretical results","venue":null,"work_id":"6f5ecb1d-54fe-4281-bdbb-8352a3ad2d19","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:a48cc18194e6c00b33db722ea32894738150335f3d6c128bc939873f464c0322","observation_id":"78f48f7a-df59-48fa-9422-db9011825e5b","resolution":{"observed_at":"2026-05-23T02:47:27.542651Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The results in 6 were produced using open-source codebases [5] [4] and models, with some small additions of code by us","venue":null,"work_id":"f1bb5ce0-895b-4156-b239-b8a2c3bcdbdf","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:e5c3742a8336573ac96e433ff839136bb2e6272477ec24d8cba7515a023fc759","observation_id":"3145bef9-d59f-475a-9721-c673f544bdd3","resolution":{"observed_at":"2026-05-23T02:47:27.514929Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"This is described in Section 5","venue":null,"work_id":"20731e47-00c6-4fb2-93e2-764a40cea446","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:e98197800325656708ab033504dbb71cb89fec95e0cd86cb2ba54f7812378342","observation_id":"a8ba68fb-4e0a-47d1-823d-1ead5b3a6ea7","resolution":{"observed_at":"2026-05-23T02:47:27.546614Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"All the other hyperparameters for training are replicated directly from the VinePPO [ 5] and Oat [4] libraries, and the user is directed to these in Section 5","venue":null,"work_id":"0dc7df98-2c3b-4a1f-b79e-6643b87ad5ca","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:d5122a18f5057a177571f5c101333de9e1a1c499c3403523c16111938e76f3a9","observation_id":"dc7ecaf9-0b14-44d1-9a44-9487efbe03b1","resolution":{"observed_at":"2026-05-23T02:47:27.560504Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"We have, however, provided training curves to aid the reader in interpreting the significance of the results","venue":null,"work_id":"844637ff-2056-418f-b638-1760d879b162","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:94ec67d92d826ed66e0af7cce70257a1a2a1e33de282ad1a4e17f5682f4bb4fa","observation_id":"d9bdbbaf-dec8-49ca-87ef-49ab1f1754ed","resolution":{"observed_at":"2026-05-23T02:47:27.614972Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"• The paper should indicate the type of compute workers CPU or GPU, internal cluster, or cloud provider, including relevant memory and storage","venue":null,"work_id":"e8418996-d16a-4999-af18-2c4176d948f2","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:6a7b6bc2c33e4fd0624f6ae808367a79512fd73821114356bc8b952cb8ecb797","observation_id":"d8e05a43-969a-4c9a-bbe6-3b08774999e8","resolution":{"observed_at":"2026-05-23T02:47:27.626385Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Guidelines: • The answer NA means that the authors have not reviewed the NeurIPS Code of Ethics","venue":null,"work_id":"fc323e45-6aa1-42d5-94cc-27d04f7b239c","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:69085096cc83a688aee9765eb6a41aec1338df249f0bbc9dcf3d7788428ca50a","observation_id":"4937820f-e32d-4279-bc46-91c07d4beddf","resolution":{"observed_at":"2026-05-23T02:47:27.538684Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Guidelines: • The answer NA means that there is no societal impact of the work performed","venue":null,"work_id":"3a5f2b2a-8a39-4c2c-93c8-1dd551023fe5","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:cccbdff6116aa9a8251d2b1315a9f5ccbcc569631b4694741961403ec315315c","observation_id":"a8a36f1f-ba68-4b49-a227-65ae6d3846c6","resolution":{"observed_at":"2026-05-23T02:47:27.589234Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Guidelines: • The answer NA means that the paper poses no such risks","venue":null,"work_id":"16d0842b-1f0d-4877-874c-f1277e7dee19","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:046d1ba87db3e68e033694a89af5afb3eb9dd10cc8a62b78b9f51caab6791c51","observation_id":"7e20c96e-748b-44cd-8914-0ed7640fc2a7","resolution":{"observed_at":"2026-05-23T02:47:27.593649Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The two libraries we used for training (VinePPO and Oat) are both fully open-source","venue":null,"work_id":"31ea4de4-36cc-4efb-8f2c-51ba27c987fa","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:01df6eb4803d2af2be48f39db515495f0dd7d0745fd16599aec5d62b5326b150","observation_id":"6f8bbaa6-8709-40d1-8dab-d2b6437daf55","resolution":{"observed_at":"2026-05-23T02:47:27.570800Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"It very simple, and could be implemented from this paper alone","venue":null,"work_id":"7f5083ca-2951-4497-ab0c-ebfb6a8ebe26","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:43b98f7412747f394f630acc65f6a3a01a6e3d2219f01787aaa440c845272dae","observation_id":"990cf32f-c5b3-4c03-90c2-0d0db59eac4a","resolution":{"observed_at":"2026-05-23T02:47:27.576905Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Guidelines: • The answer NA means that the paper does not involve crowdsourcing nor research with human subjects","venue":null,"work_id":"88de757b-5c29-477c-b32b-e15a66aaf1a2","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:7d6347af6827225ed94add1dbbc43efb49e24e75c32ee504661ad15bc9d0e9cc","observation_id":"25126526-67c4-4dd8-9def-b2d074d6cd84","resolution":{"observed_at":"2026-05-23T02:47:27.600904Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Guidelines: • The answer NA means that the paper does not involve crowdsourcing nor research with human subjects","venue":null,"work_id":"aaefd557-315d-4a57-bbfe-c97a15ef0695","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:3100342d2aa644757ae676afb0ffd413dbc72f3a476188e3b91f91e0676d8f5c","observation_id":"8aa1cc51-a145-43a9-813f-4d0b0bb8c35d","resolution":{"observed_at":"2026-05-23T02:47:27.630722Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Answer: [NA] Justification: LLM usage was only used in a standard way for editing","venue":null,"work_id":"06999c0b-724b-4e93-92f5-b974287a8c26","year":2025},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:08bef46d3a2bd966757904df01016e4baafbec4dc56181fe025449a37624de66","observation_id":"a35994fa-622d-45e2-b5de-d565cf4ea1f2","resolution":{"observed_at":"2026-05-23T02:47:27.531069Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","latest_version":6,"primary_category":"cs.LG","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability"},"reference_resolution":{"displayed":89,"state_counts":{"malformed_identifier":0,"metadata_mismatch":3,"parse_uncertain":0,"unresolved":1,"verified_exact":46,"verified_fuzzy":39},"total_outbound_references":89},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 89 of 89 outbound references and 2 inbound Pith citation observations for arXiv:2502.12272."}