{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:HHB3ZSMZ4ENUFEU2KA5KIERVBJ","short_pith_number":"pith:HHB3ZSMZ","schema_version":"1.0","canonical_sha256":"39c3bcc999e11b42929a503aa412350a6be9e0fe1cf12d9e08f1eaadb252fd71","source":{"kind":"arxiv","id":"2505.08827","version":2},"attestation_state":"computed","paper":{"title":"RLSR: Reinforcement Learning from Self Reward","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Akira Yoshiyama, Dominique Garmier, Kevin Lopez, Toby Simonds","submitted_at":"2025-05-12T23:51:04Z","abstract_excerpt":"Large language models can generate solutions to complex problems, but training them with reinforcement learning typically requires verifiable rewards that are expensive to create and not possible for all domains. We demonstrate that LLMs can effectively self-improve through self-judging without reference solutions, leveraging the inherent asymmetry between generating and verifying solutions. Our experiments show that models can provide reliable reward signals without ground truth answers, enabling reinforcement learning in domains where verifiable rewards are impractical. By implementing self-"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.08827","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-12T23:51:04Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"8479c28cfa0a570a26088c614a6295f23a077919c8be4eb7b9fa19ea3c8384de","abstract_canon_sha256":"ddb1f091996ac3666088636d3724ec257bc706b0d638007c878b5ef9a932e396"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:49:54.638934Z","signature_b64":"uV5YZDexhTn5SauNg65pGjy7JLdVocJwbRrfUiTMNo324xmC37Zp8y/gMzDqIBEPgzoBAcEq7mCjDPssw40EAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"39c3bcc999e11b42929a503aa412350a6be9e0fe1cf12d9e08f1eaadb252fd71","last_reissued_at":"2026-07-05T11:49:54.638488Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:49:54.638488Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RLSR: Reinforcement Learning from Self Reward","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Akira Yoshiyama, Dominique Garmier, Kevin Lopez, Toby Simonds","submitted_at":"2025-05-12T23:51:04Z","abstract_excerpt":"Large language models can generate solutions to complex problems, but training them with reinforcement learning typically requires verifiable rewards that are expensive to create and not possible for all domains. We demonstrate that LLMs can effectively self-improve through self-judging without reference solutions, leveraging the inherent asymmetry between generating and verifying solutions. Our experiments show that models can provide reliable reward signals without ground truth answers, enabling reinforcement learning in domains where verifiable rewards are impractical. By implementing self-"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.08827","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.08827/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.08827","created_at":"2026-07-05T11:49:54.638544+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.08827v2","created_at":"2026-07-05T11:49:54.638544+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.08827","created_at":"2026-07-05T11:49:54.638544+00:00"},{"alias_kind":"pith_short_12","alias_value":"HHB3ZSMZ4ENU","created_at":"2026-07-05T11:49:54.638544+00:00"},{"alias_kind":"pith_short_16","alias_value":"HHB3ZSMZ4ENUFEU2","created_at":"2026-07-05T11:49:54.638544+00:00"},{"alias_kind":"pith_short_8","alias_value":"HHB3ZSMZ","created_at":"2026-07-05T11:49:54.638544+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05904","citing_title":"More Convincing, Not More Correct: Self-Play Reward Hacking of Reference-Free LLM Judges","ref_index":10,"is_internal_anchor":true},{"citing_arxiv_id":"2508.19652","citing_title":"Self-Rewarding Vision-Language Model via Reasoning Decomposition","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11299","citing_title":"Primal Generation, Dual Judgment: Self-Training from Test-Time Scaling","ref_index":101,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18131","citing_title":"Training LLM Agents for Spontaneous, Reward-Free Self-Evolution via World Knowledge Exploration","ref_index":33,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HHB3ZSMZ4ENUFEU2KA5KIERVBJ","json":"https://pith.science/pith/HHB3ZSMZ4ENUFEU2KA5KIERVBJ.json","graph_json":"https://pith.science/api/pith-number/HHB3ZSMZ4ENUFEU2KA5KIERVBJ/graph.json","events_json":"https://pith.science/api/pith-number/HHB3ZSMZ4ENUFEU2KA5KIERVBJ/events.json","paper":"https://pith.science/paper/HHB3ZSMZ"},"agent_actions":{"view_html":"https://pith.science/pith/HHB3ZSMZ4ENUFEU2KA5KIERVBJ","download_json":"https://pith.science/pith/HHB3ZSMZ4ENUFEU2KA5KIERVBJ.json","view_paper":"https://pith.science/paper/HHB3ZSMZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.08827&json=true","fetch_graph":"https://pith.science/api/pith-number/HHB3ZSMZ4ENUFEU2KA5KIERVBJ/graph.json","fetch_events":"https://pith.science/api/pith-number/HHB3ZSMZ4ENUFEU2KA5KIERVBJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HHB3ZSMZ4ENUFEU2KA5KIERVBJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HHB3ZSMZ4ENUFEU2KA5KIERVBJ/action/storage_attestation","attest_author":"https://pith.science/pith/HHB3ZSMZ4ENUFEU2KA5KIERVBJ/action/author_attestation","sign_citation":"https://pith.science/pith/HHB3ZSMZ4ENUFEU2KA5KIERVBJ/action/citation_signature","submit_replication":"https://pith.science/pith/HHB3ZSMZ4ENUFEU2KA5KIERVBJ/action/replication_record"}},"created_at":"2026-07-05T11:49:54.638544+00:00","updated_at":"2026-07-05T11:49:54.638544+00:00"}