{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:PHA3Z6EKZF65DF2AZG2MN4PM5P","short_pith_number":"pith:PHA3Z6EK","schema_version":"1.0","canonical_sha256":"79c1bcf88ac97dd19740c9b4c6f1ecebe9c7aad5a7cf9d9bac69d970942a5407","source":{"kind":"arxiv","id":"2504.14363","version":2},"attestation_state":"computed","paper":{"title":"Improving RL Exploration for LLM Reasoning through Retrospective Replay","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Jingwen Xu, Muling Wu, Qi Zhang, Rui Zheng, Shihan Dou, Tao Gui, Xuanjing Huang","submitted_at":"2025-04-19T17:40:04Z","abstract_excerpt":"Reinforcement learning (RL) has increasingly become a pivotal technique in the post-training of large language models (LLMs). The effective exploration of the output space is essential for the success of RL. We observe that for complex problems, during the early stages of training, the model exhibits strong exploratory capabilities and can identify promising solution ideas. However, its limited capability at this stage prevents it from successfully solving these problems. The early suppression of these potentially valuable solution ideas by the policy gradient hinders the model's ability to re"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.14363","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-04-19T17:40:04Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"60ff9368b959d8d4ac8d7d4846ff6e5ea015bb4b10967a2bb297457a8ab9a17e","abstract_canon_sha256":"126f3583f7815d3bdffc3c12d98bac0fecf9af80fcb798ad7b8609ac929084f3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:32:28.312469Z","signature_b64":"2MRD3ON+pA1oPokajhuDQOp8/QOd+AVS4OiYNoyvd5tPW49rQT+kRry8XhClTtZxNKjtKNpOFBNAS6z0sRaMDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"79c1bcf88ac97dd19740c9b4c6f1ecebe9c7aad5a7cf9d9bac69d970942a5407","last_reissued_at":"2026-07-05T11:32:28.311981Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:32:28.311981Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Improving RL Exploration for LLM Reasoning through Retrospective Replay","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Jingwen Xu, Muling Wu, Qi Zhang, Rui Zheng, Shihan Dou, Tao Gui, Xuanjing Huang","submitted_at":"2025-04-19T17:40:04Z","abstract_excerpt":"Reinforcement learning (RL) has increasingly become a pivotal technique in the post-training of large language models (LLMs). The effective exploration of the output space is essential for the success of RL. We observe that for complex problems, during the early stages of training, the model exhibits strong exploratory capabilities and can identify promising solution ideas. However, its limited capability at this stage prevents it from successfully solving these problems. The early suppression of these potentially valuable solution ideas by the policy gradient hinders the model's ability to re"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.14363","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.14363/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.14363","created_at":"2026-07-05T11:32:28.312037+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.14363v2","created_at":"2026-07-05T11:32:28.312037+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.14363","created_at":"2026-07-05T11:32:28.312037+00:00"},{"alias_kind":"pith_short_12","alias_value":"PHA3Z6EKZF65","created_at":"2026-07-05T11:32:28.312037+00:00"},{"alias_kind":"pith_short_16","alias_value":"PHA3Z6EKZF65DF2A","created_at":"2026-07-05T11:32:28.312037+00:00"},{"alias_kind":"pith_short_8","alias_value":"PHA3Z6EK","created_at":"2026-07-05T11:32:28.312037+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21943","citing_title":"Modularized Reinforcement Learning on LLMs: From MDP Creation to Exploration and Learning","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03087","citing_title":"Learning to Solve, Forgetting to Retain: Correct-Set Turnover in RLVR","ref_index":101,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01249","citing_title":"Trust Region On-Policy Distillation","ref_index":266,"is_internal_anchor":false},{"citing_arxiv_id":"2509.08827","citing_title":"A Survey of Reinforcement Learning for Large Reasoning Models","ref_index":118,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18530","citing_title":"OGER: A Robust Offline-Guided Exploration Reward for Hybrid Reinforcement Learning","ref_index":38,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PHA3Z6EKZF65DF2AZG2MN4PM5P","json":"https://pith.science/pith/PHA3Z6EKZF65DF2AZG2MN4PM5P.json","graph_json":"https://pith.science/api/pith-number/PHA3Z6EKZF65DF2AZG2MN4PM5P/graph.json","events_json":"https://pith.science/api/pith-number/PHA3Z6EKZF65DF2AZG2MN4PM5P/events.json","paper":"https://pith.science/paper/PHA3Z6EK"},"agent_actions":{"view_html":"https://pith.science/pith/PHA3Z6EKZF65DF2AZG2MN4PM5P","download_json":"https://pith.science/pith/PHA3Z6EKZF65DF2AZG2MN4PM5P.json","view_paper":"https://pith.science/paper/PHA3Z6EK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.14363&json=true","fetch_graph":"https://pith.science/api/pith-number/PHA3Z6EKZF65DF2AZG2MN4PM5P/graph.json","fetch_events":"https://pith.science/api/pith-number/PHA3Z6EKZF65DF2AZG2MN4PM5P/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PHA3Z6EKZF65DF2AZG2MN4PM5P/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PHA3Z6EKZF65DF2AZG2MN4PM5P/action/storage_attestation","attest_author":"https://pith.science/pith/PHA3Z6EKZF65DF2AZG2MN4PM5P/action/author_attestation","sign_citation":"https://pith.science/pith/PHA3Z6EKZF65DF2AZG2MN4PM5P/action/citation_signature","submit_replication":"https://pith.science/pith/PHA3Z6EKZF65DF2AZG2MN4PM5P/action/replication_record"}},"created_at":"2026-07-05T11:32:28.312037+00:00","updated_at":"2026-07-05T11:32:28.312037+00:00"}