{"as_of":"2026-08-09T06:56:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:fc4fe72df488238be91908ba5f97d127c7512f5e8bc956b36ad7bb6ec6cf9dcb","coverage":[{"denominator":38,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":38,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T15:11:08.217318Z","state":"measured"},{"denominator":48,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":48,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":10,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":10,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T20:26:01.278862Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-22T11:41:29.623621Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"cited_work":{"arxiv_id":"2502.06533","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.06533","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Vassoyan, J., Beau, N., and Plaud, R","venue":null,"work_id":"085cbb78-7060-495d-85d0-e20cc389840b","year":2025},"citing_paper":{"arxiv_id":"2506.01939","last_updated":"2025-11-13T10:08:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-02T17:54:39Z","title":"Beyond the 80/20 Rule: High-Entropy Minority Tokens Drive Effective Reinforcement Learning for LLM Reasoning","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-12T12:12:08.724844Z"},"links":{"cited_paper":"/paper/2502.06533","citing_paper":"/paper/2506.01939"},"observation_digest":"sha256:b868a2b7a8a0876369af1b4fa43651c6f9f8f8ab80bf2009c8b332da2e598857","observation_id":"7674cb9a-22f7-4bfd-b688-55d4dd1f049a","resolution":{"observed_at":"2026-05-12T12:12:08.960644Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.06533","snapshot_observed_at":"2026-08-06T20:26:01.278862Z","title":"Vassoyan, N","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.03190","last_updated":"2025-07-11T16:55:14Z","snapshot_observed_at":"2026-08-06T20:13:48.510141Z","submitted_at":"2025-07-03T21:45:17Z","title":"Discovering Algorithms with Computational Language Processing","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T20:26:01.278862Z"},"links":{"cited_paper":"/paper/2502.06533","citing_paper":"/paper/2507.03190"},"observation_digest":"sha256:e9c7d60e52b91b9d85dd0ecd49fa87021def7729d0d3882267cc631d81f26230","observation_id":"44b0d37e-a9a3-4184-9cb0-6a9915ea55e8","resolution":{"observed_at":"2026-08-06T20:26:01.278862Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.06533","snapshot_observed_at":"2026-08-06T19:18:38.761833Z","title":"Ignore the kl penalty! boosting exploration on critical tokens to enhance rl fine-tuning.arXiv preprint arXiv:2502.06533, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.06013","last_updated":"2025-07-08T14:17:07Z","snapshot_observed_at":"2026-08-07T22:57:58.555368Z","submitted_at":"2025-07-08T14:17:07Z","title":"CogniSQL-R1-Zero: Lightweight Reinforced Reasoning for Efficient SQL Generation","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T19:18:38.761833Z"},"links":{"cited_paper":"/paper/2502.06533","citing_paper":"/paper/2507.06013"},"observation_digest":"sha256:51314340d4d3b8853f3f7d1dde2ecb27cb8a7c3860ea0030085992f3bee8a351","observation_id":"6a1c392c-5e2c-4a08-a44c-b3dcb06b6e52","resolution":{"observed_at":"2026-08-06T19:18:38.761833Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"cited_work":{"arxiv_id":"2502.06533","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.06533","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Vassoyan, J., Beau, N., and Plaud, R","venue":null,"work_id":"085cbb78-7060-495d-85d0-e20cc389840b","year":2025},"citing_paper":{"arxiv_id":"2507.15778","last_updated":"2026-05-15T04:36:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T16:34:01Z","title":"Stabilizing Knowledge, Promoting Reasoning: Dual-Token Constraints for RLVR","version":2},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-05-21T23:20:45.685446Z"},"links":{"cited_paper":"/paper/2502.06533","citing_paper":"/paper/2507.15778"},"observation_digest":"sha256:407295b587fa41f421a8191f35939fbc41e384340d9af53da8d5e0d0eabb32c4","observation_id":"1eab7712-b1b2-44ff-98fa-4910f7e8c24c","resolution":{"observed_at":"2026-05-21T23:24:26.241888Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.06533","snapshot_observed_at":"2026-08-04T12:41:43.057152Z","title":"Ignore the kl penalty! boosting exploration on critical tokens to enhance rl fine-tuning.arXiv preprint arXiv:2502.06533,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2510.02919","last_updated":"2026-05-29T16:16:27Z","snapshot_observed_at":"2026-08-07T08:42:12.238625Z","submitted_at":"2025-10-03T11:46:04Z","title":"Self-Reflective Generation at Test Time","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-04T12:41:43.057152Z"},"links":{"cited_paper":"/paper/2502.06533","citing_paper":"/paper/2510.02919"},"observation_digest":"sha256:126fb908206a3d1e58bdba5e63e8ed8f440e113ddf3f29a3bc7c668db2384b50","observation_id":"376af9d9-ebde-4618-9197-5700c26f2387","resolution":{"observed_at":"2026-08-04T12:41:43.057152Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.06533","snapshot_observed_at":"2026-08-04T09:48:09.263325Z","title":"Ignore the kl penalty! boosting exploration on critical tokens to enhance rl fine-tuning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2510.13554","last_updated":"2026-06-08T06:02:10Z","snapshot_observed_at":"2026-08-04T09:47:55.188440Z","submitted_at":"2025-10-15T13:49:51Z","title":"Attention Illuminates LLM Reasoning: The Preplan-and-Anchor Rhythm Enables Fine-Grained Policy Optimization","version":2},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-04T09:48:09.263325Z"},"links":{"cited_paper":"/paper/2502.06533","citing_paper":"/paper/2510.13554"},"observation_digest":"sha256:b0b51152c3b4a6a7393517a466c4e4483a59f7aa5c9df5b36113173bad5b29eb","observation_id":"0b57fbfe-e87d-4eb8-bf7b-53889a8b50a3","resolution":{"observed_at":"2026-08-04T09:48:09.263325Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"cited_work":{"arxiv_id":"2502.06533","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.06533","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Vassoyan, J., Beau, N., and Plaud, R","venue":null,"work_id":"085cbb78-7060-495d-85d0-e20cc389840b","year":2025},"citing_paper":{"arxiv_id":"2601.10348","last_updated":"2026-05-21T06:29:24Z","snapshot_observed_at":"2026-08-03T01:57:39.427752Z","submitted_at":"2026-01-15T12:45:05Z","title":"Training-Trajectory-Aware Token Selection","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-22T11:41:21.275802Z"},"links":{"cited_paper":"/paper/2502.06533","citing_paper":"/paper/2601.10348"},"observation_digest":"sha256:267d7f5e1b7726bc56d964f44e357334cd8d91101631d9831cceddf770570f01","observation_id":"68e48558-ca96-4c19-b535-78a32c01ac09","resolution":{"observed_at":"2026-05-22T11:41:29.626509Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.06533","snapshot_observed_at":"2026-07-13T14:33:35.834383Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical T okens to Enhance RL Fine-T uning.CoRR, abs/2502.06533,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2604.01193","last_updated":"2026-06-24T22:44:36Z","snapshot_observed_at":"2026-08-02T16:28:38.685624Z","submitted_at":"2026-04-01T17:39:50Z","title":"Embarrassingly Simple Self-Distillation Improves Code Generation","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-07-13T14:33:35.834383Z"},"links":{"cited_paper":"/paper/2502.06533","citing_paper":"/paper/2604.01193"},"observation_digest":"sha256:d4a2ec4923e7c9ff16ac86bf53c2c5bec49a63c2f2e93d32acba318f36369edf","observation_id":"713e4f03-97af-4015-a2a7-e47620c85e4b","resolution":{"observed_at":"2026-07-13T14:33:35.834383Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"cited_work":{"arxiv_id":"2502.06533","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.06533","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Vassoyan, J., Beau, N., and Plaud, R","venue":null,"work_id":"085cbb78-7060-495d-85d0-e20cc389840b","year":2025},"citing_paper":{"arxiv_id":"2604.16158","last_updated":"2026-04-17T15:27:35Z","snapshot_observed_at":"2026-08-02T08:13:50.306987Z","submitted_at":"2026-04-17T15:27:35Z","title":"AtManRL: Towards Faithful Reasoning via Differentiable Attention Saliency","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-10T08:08:50.330857Z"},"links":{"cited_paper":"/paper/2502.06533","citing_paper":"/paper/2604.16158"},"observation_digest":"sha256:762cce10ceef8c668961782af6f45c0b1d49d7452ab80a98973ac0289333c1f2","observation_id":"9ccb437f-a756-4f65-9150-38507645632c","resolution":{"observed_at":"2026-05-10T08:22:37.709068Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"cited_work":{"arxiv_id":"2502.06533","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.06533","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Vassoyan, J., Beau, N., and Plaud, R","venue":null,"work_id":"085cbb78-7060-495d-85d0-e20cc389840b","year":2025},"citing_paper":{"arxiv_id":"2605.08283","last_updated":"2026-05-08T07:38:35Z","snapshot_observed_at":"2026-08-03T16:22:22.745568Z","submitted_at":"2026-05-08T07:38:35Z","title":"HTPO: Towards Exploration-Exploitation Balanced Policy Optimization via Hierarchical Token-level Objective Control","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-12T00:50:42.836549Z"},"links":{"cited_paper":"/paper/2502.06533","citing_paper":"/paper/2605.08283"},"observation_digest":"sha256:96e1e35838c27b95e54dbdab5885c4e8188c4e99ba3ec3fd7fefae9468f5cc21","observation_id":"3164a8ee-072f-4e86-80d7-3c6c4139884a","resolution":{"observed_at":"2026-05-12T00:51:14.645501Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2502.06533/citation-record","integrity":"/paper/2502.06533/integrity","json":"/paper/2502.06533/citation-record.json","paper":"/paper/2502.06533"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:11:08.595692Z","title":null,"venue":null,"work_id":"6c6105ed-b0a3-4e79-822f-983787f9e692","year":2022},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.108360Z"},"links":{"citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:ea423c81d02fab12f38671f3a813b487186849e96ac1f199dee3376c8780310e","observation_id":"4335a7de-95b4-4289-beb7-31d9e2830a4a","resolution":{"observed_at":"2026-08-08T15:11:08.598573Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:11:08.588340Z","title":null,"venue":null,"work_id":"6e9a9988-b497-4727-8932-62c526fa7ee4","year":2024},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.111751Z"},"links":{"citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:64d6052892b56aa3a2b48763012dab5cc6e9b1ab20995df8299c47d998f13016","observation_id":"2be403d8-dfcc-4cbe-acd2-a0877eae64d9","resolution":{"observed_at":"2026-08-08T15:11:08.590593Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:11:08.579898Z","title":null,"venue":null,"work_id":"51611c81-70e7-4010-ae2b-5fb6678a1fba","year":1953},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.114846Z"},"links":{"citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:04cbf7776bb8aefb48fe7bc25ae199ba6c9372bed515cb58cc181a0380fcb04b","observation_id":"cb02443b-5e7c-454d-80a1-cdaf2f15c3f6","resolution":{"observed_at":"2026-08-08T15:11:08.583473Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:11:08.119207Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.119207Z"},"links":{"citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:98162c0c0032f8f73c8bef06127f2e35f11dfdc1f83291bfd453a0d480a84ca5","observation_id":"f5e03d1f-0a1a-4099-98da-8e6dfe7a5b76","resolution":{"observed_at":"2026-08-08T15:11:08.119207Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:11:08.572294Z","title":null,"venue":null,"work_id":"b37e777a-612a-4c42-9c83-a5c1493b1858","year":2023},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.122210Z"},"links":{"citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:1e6bb2b4b1618ccc9a7b93fc8735b8b485eb9509cc4faba6b62ef3fbbf6a1a07","observation_id":"abae6b32-0469-4597-84ce-2779d1d152b1","resolution":{"observed_at":"2026-08-08T15:11:08.574958Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2107.03374","last_updated":"2021-07-14T17:16:02Z","snapshot_observed_at":"2026-08-08T11:58:24.516369Z","submitted_at":"2021-07-07T17:41:24Z","title":"Evaluating Large Language Models Trained on Code","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2107.03374","snapshot_observed_at":"2026-08-08T15:11:08.125025Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.125025Z"},"links":{"cited_paper":"/paper/2107.03374","citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:2fb0e05ed626e07c5bff9bda93b26cb64fce246e2d44a4fd2404d21d08a4187e","observation_id":"918f833d-6546-4426-aabd-16754a3d7bdd","resolution":{"observed_at":"2026-08-08T15:11:08.125025Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:11:08.564166Z","title":null,"venue":null,"work_id":"a140a0f1-12b2-4a1a-aa3a-114e95dd9e5b","year":2017},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.128531Z"},"links":{"citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:70f259b195fead93fead44cdf2c2e10f0cdc7f4272ac881bc85f6e056e222c7a","observation_id":"a65dbc7e-62ff-433b-8364-619b03875d86","resolution":{"observed_at":"2026-08-08T15:11:08.567131Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-07T01:45:38.840969Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-08T15:11:08.131846Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.131846Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:7d8be31661594b34aa995fe3c25d22088639b382c85c194ed0c571598a605a3d","observation_id":"3db7b570-de64-4666-901b-8d936ac93fe0","resolution":{"observed_at":"2026-08-08T15:11:08.131846Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-08T15:11:08.134977Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.134977Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:6d70fc37232fbc4c42993730c9322115e6cab16630fa2ab78649dcd2db37d19d","observation_id":"c368a59c-9816-4431-9826-845747dd1018","resolution":{"observed_at":"2026-08-08T15:11:08.134977Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04642","last_updated":"2024-03-07T16:36:29Z","snapshot_observed_at":"2026-08-07T01:30:54.327974Z","submitted_at":"2024-03-07T16:36:29Z","title":"Teaching Large Language Models to Reason with Reinforcement Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04642","snapshot_observed_at":"2026-08-08T15:11:08.137792Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.137792Z"},"links":{"cited_paper":"/paper/2403.04642","citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:82ff676c8f1e813cf1784bbe3e4687bdd74c428939c2c4240338c79e76ce0d73","observation_id":"0d0b062e-1036-4c1f-b7fe-403edeb2d6d2","resolution":{"observed_at":"2026-08-08T15:11:08.137792Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:11:08.140623Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.140623Z"},"links":{"citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:9f3819c6579faf28e51b6da24fda384fcc7f336234873f7312d28211909e118b","observation_id":"ab25b46b-525a-4587-ac65-4057203baa0b","resolution":{"observed_at":"2026-08-08T15:11:08.140623Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:11:08.552624Z","title":null,"venue":null,"work_id":"8aeb4a7d-69c3-4afa-a812-65cda7e4f1e5","year":2024},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.143465Z"},"links":{"citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:64b5a5ef2f515c5652b90280dced2de28dfb49f6bf828d84ed396586fc2cbac6","observation_id":"fabb473f-eb47-46c3-a79e-e01d1d4da8f1","resolution":{"observed_at":"2026-08-08T15:11:08.555008Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:11:08.545586Z","title":null,"venue":null,"work_id":"e5f84858-015c-41ab-8027-c402dd8b4f1e","year":2023},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.145539Z"},"links":{"citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:ffe1113fc7ffea6a884fa9d5bf835a0823767b61dfa507f3ef7d38a977c22f9c","observation_id":"ef6d4f98-1704-467a-a0a5-21484918da71","resolution":{"observed_at":"2026-08-08T15:11:08.547831Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:11:08.147904Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.147904Z"},"links":{"citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:2623b87633d0b53d6202d270832d7dd8fe08642b9a632e8fc234b75e402e07d1","observation_id":"f1b2ed8f-87c8-44a9-ade0-4021d5e97acd","resolution":{"observed_at":"2026-08-08T15:11:08.147904Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:11:08.539054Z","title":"Lee, Kangwook Lee, and Dimitris Papailiopoulos","venue":null,"work_id":"e9f62d3e-d30f-4ef5-ac52-cbbf633b06b7","year":2024},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.150117Z"},"links":{"citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:2c362fc49b9b547b3b9de710626cbad836e4ae2263b47701a6e06ee3778736b4","observation_id":"6086b97c-1a84-413e-a8a6-fa899a26aca7","resolution":{"observed_at":"2026-08-08T15:11:08.541335Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:11:08.152169Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.152169Z"},"links":{"citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:fe0f645eb97e6da85d514c484127c021a72ab54f1c8aabb01e63fa63390716be","observation_id":"028bfb75-43f9-46a5-9400-58b7aba0a563","resolution":{"observed_at":"2026-08-08T15:11:08.152169Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.14201","last_updated":"2023-05-23T16:20:30Z","snapshot_observed_at":"2026-07-06T15:31:36.425535Z","submitted_at":"2023-05-23T16:20:30Z","title":"Goat: Fine-tuned LLaMA Outperforms GPT-4 on Arithmetic Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.14201","snapshot_observed_at":"2026-08-08T15:11:08.154302Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.154302Z"},"links":{"cited_paper":"/paper/2305.14201","citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:cf123aff5fd88804a4cc6add012b0925c834bd675fdc97aa54aafe9ded1d61a1","observation_id":"4389c21f-a67e-4a21-8939-300e35901201","resolution":{"observed_at":"2026-08-08T15:11:08.154302Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:11:08.531111Z","title":"Bartoldson, Bhavya Kailkhura, Abhinav Bhatele, Jonas Geiping, Avi Schwarzschild, and Tom Goldstein","venue":null,"work_id":"684baef1-bb0b-40c0-ac9c-4efe74e2ac0a","year":2024},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.157948Z"},"links":{"citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:4db2d20ba50579573df101ef288a9882f538abdb2006d23f283ea229434602aa","observation_id":"92dfa1ca-53fb-45a5-93cf-af728f0d6dee","resolution":{"observed_at":"2026-08-08T15:11:08.534555Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:11:08.523536Z","title":null,"venue":null,"work_id":"8263c37d-5dec-4b91-9c24-e5505a089411","year":2016},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.160447Z"},"links":{"citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:6ad10a2617f27528524cda281448ba003db9760a654845ad3085423b6755229c","observation_id":"a0a11701-511c-4602-a25e-807baca51cc0","resolution":{"observed_at":"2026-08-08T15:11:08.526289Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:11:08.515642Z","title":null,"venue":null,"work_id":"a91d46b6-f831-4937-95b4-a4a9b012ce72","year":2023},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.163134Z"},"links":{"citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:6d83863ede34bb952e96aa9a99ebbdd4404a80ca17b1ff10929cc9bf6a3f73f5","observation_id":"020d0dcb-a310-4071-aa29-30b33186f5fa","resolution":{"observed_at":"2026-08-08T15:11:08.518595Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:11:08.165824Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.165824Z"},"links":{"citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:00e76774ca55d6a134b46fb0ad4a424e263951f920f39a55710ed1bd7b84721f","observation_id":"24008427-3b92-4768-b228-d42bd07c16e4","resolution":{"observed_at":"2026-08-08T15:11:08.165824Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:11:08.503252Z","title":null,"venue":null,"work_id":"e40bdbb8-95e8-41ef-bceb-32e028ce085b","year":2020},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.168861Z"},"links":{"citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:0827548e24cbe611ab8356920b7c7fcb84849b1eb748a48ff6eb3d41c49397fd","observation_id":"757c0598-cfca-444c-9323-03d517276ec4","resolution":{"observed_at":"2026-08-08T15:11:08.506127Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.14737","last_updated":"2023-11-22T00:31:01Z","snapshot_observed_at":"2026-07-06T16:52:03.370519Z","submitted_at":"2023-11-22T00:31:01Z","title":"Positional Description Matters for Transformers Arithmetic","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.14737","snapshot_observed_at":"2026-08-08T15:11:08.172446Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.172446Z"},"links":{"cited_paper":"/paper/2311.14737","citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:8a83d41da074879f238b8e96057bc30f0c1823c9f877274ce600994b8682fb1f","observation_id":"c9a8232f-e00c-44de-8223-7c30eedb9ef1","resolution":{"observed_at":"2026-08-08T15:11:08.172446Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:11:08.495488Z","title":null,"venue":null,"work_id":"8ba57a73-c137-4e92-b337-d2781ffdc2a1","year":2020},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.176269Z"},"links":{"citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:791bfc829cce0e0e78b4107d7fff17f7255288ca0fdb353c044587cc228d02cb","observation_id":"cac9a057-a2aa-416f-9395-58ec1ee3f4d2","resolution":{"observed_at":"2026-08-08T15:11:08.498250Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-08T15:11:08.179605Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.179605Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:3a9d086a0040025c6e3d719138d04d75bd217401ea9e4b4e19ee91bdca7d6cd7","observation_id":"55a321ac-facf-48ee-b2ac-08c9b019dc82","resolution":{"observed_at":"2026-08-08T15:11:08.179605Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:11:08.182556Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.182556Z"},"links":{"citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:cef1cf675726ab2956c64edd4900fc1ff2b2b2fff486984d8229c43a53b68349","observation_id":"24c77f65-7e3f-4ac4-acfc-a7da8bfd6b90","resolution":{"observed_at":"2026-08-08T15:11:08.182556Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:11:08.185354Z","title":"Chi, Quoc V","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.185354Z"},"links":{"citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:3a33b3d680a33e567acbba8b802a4df12dea94486b9830781a32fba46c56aac3","observation_id":"636c5415-2dd8-4bb2-a643-7afc60724c4b","resolution":{"observed_at":"2026-08-08T15:11:08.185354Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:11:08.483919Z","title":null,"venue":null,"work_id":"3547e445-4b42-4eba-b376-51b27af3d129","year":2022},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.188614Z"},"links":{"citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:c295fa56991db6e683bb3f22e0b35a6f507b785f0f332c711659e55ed9119152","observation_id":"665bf2e8-fd54-4a58-9104-00a2bd0f3aaa","resolution":{"observed_at":"2026-08-08T15:11:08.486457Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:11:08.191550Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.191550Z"},"links":{"citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:86453afe7ab3b8de5a1c6f07b4b9b7cae3e4093f7b8c6c5740fc25e1a709821b","observation_id":"0515158b-359e-4933-9066-0b29407c7474","resolution":{"observed_at":"2026-08-08T15:11:08.191550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16173","last_updated":"2023-12-06T16:31:50Z","snapshot_observed_at":"2026-07-06T16:53:22.999597Z","submitted_at":"2023-11-22T03:36:18Z","title":"Conditions for Length Generalization in Learning Reasoning Skills","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.16173","snapshot_observed_at":"2026-08-08T15:11:08.194619Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.194619Z"},"links":{"cited_paper":"/paper/2311.16173","citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:d2313219c02aa6eb7aafbe4bb1bcb55970bebcf8ae07e397d7a5ad94023ca514","observation_id":"cde243c6-e19a-47f4-9383-71f9af953329","resolution":{"observed_at":"2026-08-08T15:11:08.194619Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:11:08.197539Z","title":"Hausknecht, and Karthik Narasimhan","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.197539Z"},"links":{"citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:a5070c06de56c74d01575db1da0cce4c99d7c3413404bd38b1f526d8dca39449","observation_id":"d5e1c47b-f1a6-4b0f-9def-eda2b3f42635","resolution":{"observed_at":"2026-08-08T15:11:08.197539Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.02015","last_updated":"2023-03-16T09:28:15Z","snapshot_observed_at":"2026-08-04T17:38:55.806742Z","submitted_at":"2023-03-16T09:28:15Z","title":"How well do Large Language Models perform in Arithmetic tasks?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.02015","snapshot_observed_at":"2026-08-08T15:11:08.200325Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.200325Z"},"links":{"cited_paper":"/paper/2304.02015","citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:fcbcb7ca28240f62d79f967ed81bcea3cea03282b64c67227e45c621dff3ab10","observation_id":"cdf1b3b7-0ef0-4a28-98f7-d25e3ee8bb00","resolution":{"observed_at":"2026-08-08T15:11:08.200325Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:11:08.472202Z","title":null,"venue":null,"work_id":"a8eeb45d-e224-4e15-b009-8cbded6343af","year":2024},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.203740Z"},"links":{"citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:8bf506bcabe34da8d0144c12bc8e20504302f123031596a039cd977ab88249e8","observation_id":"6c344090-0431-4f4e-add7-abf0ab2e5722","resolution":{"observed_at":"2026-08-08T15:11:08.474833Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.08589","last_updated":"2023-11-08T18:10:10Z","snapshot_observed_at":"2026-08-05T19:09:04.764238Z","submitted_at":"2023-09-15T17:44:17Z","title":"Chain-of-Thought Reasoning is a Policy Improvement Operator","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.08589","snapshot_observed_at":"2026-08-08T15:11:08.206499Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.206499Z"},"links":{"cited_paper":"/paper/2309.08589","citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:a05d6d73d5141f570d521ec1bea6ac76bbc7f94f6c95065b3ff4bff4a87be55c","observation_id":"18beb684-bb45-4677-b7c7-b82c595574a7","resolution":{"observed_at":"2026-08-08T15:11:08.206499Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:11:08.461694Z","title":null,"venue":null,"work_id":"48c92fcb-cb61-4f03-a07d-e18520127f3c","year":2024},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.209904Z"},"links":{"citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:3330b6dcaf288de9d096ce8c742415d476fec47b00eda07775306cdf15a743f0","observation_id":"c16742db-1a7f-40b6-9c3a-95ea4da776e1","resolution":{"observed_at":"2026-08-08T15:11:08.466799Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.08593","last_updated":"2020-01-08T23:02:36Z","snapshot_observed_at":"2026-08-08T07:44:49.921146Z","submitted_at":"2019-09-18T17:33:39Z","title":"Fine-Tuning Language Models from Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.08593","snapshot_observed_at":"2026-08-08T15:11:08.212092Z","title":"Ziegler, Nisan Stiennon, Jeffrey Wu, Tom B","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.212092Z"},"links":{"cited_paper":"/paper/1909.08593","citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:87ab43bf4184b0d0bddc5a6fd94cecbf71baf5f9c3dfb2fc8df4970c464f5ef8","observation_id":"9c541d36-6aa4-4a9a-a160-5566293ec96d","resolution":{"observed_at":"2026-08-08T15:11:08.212092Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:11:08.214627Z","title":"online\" 'onlinestring :=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.214627Z"},"links":{"citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:2d0e4457b9b601f164285a4bc867a6ea955bd089e3975d06b9b8f54987c40b35","observation_id":"dfd56579-ae4d-4fdf-b3ae-cb16ce856c98","resolution":{"observed_at":"2026-08-08T15:11:08.214627Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:11:08.217318Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-08T15:11:08.217318Z"},"links":{"citing_paper":"/paper/2502.06533"},"observation_digest":"sha256:38aa0683b5784e805d1c96ca1845de9efee1f632214622d752034d06bdd66299","observation_id":"453d04b3-385b-43cc-b44c-fb4d7f650d30","resolution":{"observed_at":"2026-08-08T15:11:08.217318Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2502.06533","last_updated":"2025-02-10T14:56:25Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-08T15:05:57.951816Z","submitted_at":"2025-02-10T14:56:25Z","title":"Ignore the KL Penalty! Boosting Exploration on Critical Tokens to Enhance RL Fine-Tuning"},"reference_resolution":{"displayed":38,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":36,"verified_exact":0,"verified_fuzzy":2},"total_outbound_references":38},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 38 of 38 outbound references and 10 inbound Pith citation observations for arXiv:2502.06533."}