{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:PHA3Z6EKZF65DF2AZG2MN4PM5P","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"126f3583f7815d3bdffc3c12d98bac0fecf9af80fcb798ad7b8609ac929084f3","cross_cats_sorted":["cs.CL"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-04-19T17:40:04Z","title_canon_sha256":"60ff9368b959d8d4ac8d7d4846ff6e5ea015bb4b10967a2bb297457a8ab9a17e"},"schema_version":"1.0","source":{"id":"2504.14363","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2504.14363","created_at":"2026-07-05T11:32:28Z"},{"alias_kind":"arxiv_version","alias_value":"2504.14363v2","created_at":"2026-07-05T11:32:28Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.14363","created_at":"2026-07-05T11:32:28Z"},{"alias_kind":"pith_short_12","alias_value":"PHA3Z6EKZF65","created_at":"2026-07-05T11:32:28Z"},{"alias_kind":"pith_short_16","alias_value":"PHA3Z6EKZF65DF2A","created_at":"2026-07-05T11:32:28Z"},{"alias_kind":"pith_short_8","alias_value":"PHA3Z6EK","created_at":"2026-07-05T11:32:28Z"}],"graph_snapshots":[{"event_id":"sha256:97aadc8c045d05fd2a185fa15720f48913b319fbc1afb63da0459bc2a42c60fa","target":"graph","created_at":"2026-07-05T11:32:28Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2504.14363/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning (RL) has increasingly become a pivotal technique in the post-training of large language models (LLMs). The effective exploration of the output space is essential for the success of RL. We observe that for complex problems, during the early stages of training, the model exhibits strong exploratory capabilities and can identify promising solution ideas. However, its limited capability at this stage prevents it from successfully solving these problems. The early suppression of these potentially valuable solution ideas by the policy gradient hinders the model's ability to re","authors_text":"Jingwen Xu, Muling Wu, Qi Zhang, Rui Zheng, Shihan Dou, Tao Gui, Xuanjing Huang","cross_cats":["cs.CL"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-04-19T17:40:04Z","title":"Improving RL Exploration for LLM Reasoning through Retrospective Replay"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.14363","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:ce00a700c288848e62bbbee17b268fc83f7910d502ed7b802352afb05a394af1","target":"record","created_at":"2026-07-05T11:32:28Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"126f3583f7815d3bdffc3c12d98bac0fecf9af80fcb798ad7b8609ac929084f3","cross_cats_sorted":["cs.CL"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-04-19T17:40:04Z","title_canon_sha256":"60ff9368b959d8d4ac8d7d4846ff6e5ea015bb4b10967a2bb297457a8ab9a17e"},"schema_version":"1.0","source":{"id":"2504.14363","kind":"arxiv","version":2}},"canonical_sha256":"79c1bcf88ac97dd19740c9b4c6f1ecebe9c7aad5a7cf9d9bac69d970942a5407","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"79c1bcf88ac97dd19740c9b4c6f1ecebe9c7aad5a7cf9d9bac69d970942a5407","first_computed_at":"2026-07-05T11:32:28.311981Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:32:28.311981Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"2MRD3ON+pA1oPokajhuDQOp8/QOd+AVS4OiYNoyvd5tPW49rQT+kRry8XhClTtZxNKjtKNpOFBNAS6z0sRaMDQ==","signature_status":"signed_v1","signed_at":"2026-07-05T11:32:28.312469Z","signed_message":"canonical_sha256_bytes"},"source_id":"2504.14363","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:ce00a700c288848e62bbbee17b268fc83f7910d502ed7b802352afb05a394af1","sha256:97aadc8c045d05fd2a185fa15720f48913b319fbc1afb63da0459bc2a42c60fa"],"state_sha256":"13ec825707944e34831c4a68919ff3c7da97f384f2275b3882ce450a06cbad25"}