{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:HHB3ZSMZ4ENUFEU2KA5KIERVBJ","short_pith_number":"pith:HHB3ZSMZ","canonical_record":{"source":{"id":"2505.08827","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-12T23:51:04Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"8479c28cfa0a570a26088c614a6295f23a077919c8be4eb7b9fa19ea3c8384de","abstract_canon_sha256":"ddb1f091996ac3666088636d3724ec257bc706b0d638007c878b5ef9a932e396"},"schema_version":"1.0"},"canonical_sha256":"39c3bcc999e11b42929a503aa412350a6be9e0fe1cf12d9e08f1eaadb252fd71","source":{"kind":"arxiv","id":"2505.08827","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2505.08827","created_at":"2026-07-05T11:49:54Z"},{"alias_kind":"arxiv_version","alias_value":"2505.08827v2","created_at":"2026-07-05T11:49:54Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.08827","created_at":"2026-07-05T11:49:54Z"},{"alias_kind":"pith_short_12","alias_value":"HHB3ZSMZ4ENU","created_at":"2026-07-05T11:49:54Z"},{"alias_kind":"pith_short_16","alias_value":"HHB3ZSMZ4ENUFEU2","created_at":"2026-07-05T11:49:54Z"},{"alias_kind":"pith_short_8","alias_value":"HHB3ZSMZ","created_at":"2026-07-05T11:49:54Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:HHB3ZSMZ4ENUFEU2KA5KIERVBJ","target":"record","payload":{"canonical_record":{"source":{"id":"2505.08827","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-12T23:51:04Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"8479c28cfa0a570a26088c614a6295f23a077919c8be4eb7b9fa19ea3c8384de","abstract_canon_sha256":"ddb1f091996ac3666088636d3724ec257bc706b0d638007c878b5ef9a932e396"},"schema_version":"1.0"},"canonical_sha256":"39c3bcc999e11b42929a503aa412350a6be9e0fe1cf12d9e08f1eaadb252fd71","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:49:54.638934Z","signature_b64":"uV5YZDexhTn5SauNg65pGjy7JLdVocJwbRrfUiTMNo324xmC37Zp8y/gMzDqIBEPgzoBAcEq7mCjDPssw40EAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"39c3bcc999e11b42929a503aa412350a6be9e0fe1cf12d9e08f1eaadb252fd71","last_reissued_at":"2026-07-05T11:49:54.638488Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:49:54.638488Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2505.08827","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:49:54Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"se9k8vZkOlZaxEIwWvJc87qbV+D6kubhK24u5OG4YSuYszpkmSZVUB/J7rJXMz/ToDaw6oHIYhueKiYDP8Q1Aw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-15T04:50:23.113239Z"},"content_sha256":"cfe80d7750efc2b0d8e638fce18e8213e59a769ae6e642997d74505013197d12","schema_version":"1.0","event_id":"sha256:cfe80d7750efc2b0d8e638fce18e8213e59a769ae6e642997d74505013197d12"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:HHB3ZSMZ4ENUFEU2KA5KIERVBJ","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"RLSR: Reinforcement Learning from Self Reward","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Akira Yoshiyama, Dominique Garmier, Kevin Lopez, Toby Simonds","submitted_at":"2025-05-12T23:51:04Z","abstract_excerpt":"Large language models can generate solutions to complex problems, but training them with reinforcement learning typically requires verifiable rewards that are expensive to create and not possible for all domains. We demonstrate that LLMs can effectively self-improve through self-judging without reference solutions, leveraging the inherent asymmetry between generating and verifying solutions. Our experiments show that models can provide reliable reward signals without ground truth answers, enabling reinforcement learning in domains where verifiable rewards are impractical. By implementing self-"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.08827","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.08827/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:49:54Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"WtWkaXLCh6YGtjnT2slfw3Ejv11GsuJcObXnBz64qqHwOTPKwabdY5oE+phPLXdtghJJBtmBmIyf/vG3bFrgBw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-15T04:50:23.114006Z"},"content_sha256":"39c1f47c43edb4bedbac0fc274594759b7eed2ba166091e1f47bffcc4c745ab4","schema_version":"1.0","event_id":"sha256:39c1f47c43edb4bedbac0fc274594759b7eed2ba166091e1f47bffcc4c745ab4"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/HHB3ZSMZ4ENUFEU2KA5KIERVBJ/bundle.json","state_url":"https://pith.science/pith/HHB3ZSMZ4ENUFEU2KA5KIERVBJ/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/HHB3ZSMZ4ENUFEU2KA5KIERVBJ/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-15T04:50:23Z","links":{"resolver":"https://pith.science/pith/HHB3ZSMZ4ENUFEU2KA5KIERVBJ","bundle":"https://pith.science/pith/HHB3ZSMZ4ENUFEU2KA5KIERVBJ/bundle.json","state":"https://pith.science/pith/HHB3ZSMZ4ENUFEU2KA5KIERVBJ/state.json","well_known_bundle":"https://pith.science/.well-known/pith/HHB3ZSMZ4ENUFEU2KA5KIERVBJ/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:HHB3ZSMZ4ENUFEU2KA5KIERVBJ","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"ddb1f091996ac3666088636d3724ec257bc706b0d638007c878b5ef9a932e396","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-12T23:51:04Z","title_canon_sha256":"8479c28cfa0a570a26088c614a6295f23a077919c8be4eb7b9fa19ea3c8384de"},"schema_version":"1.0","source":{"id":"2505.08827","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2505.08827","created_at":"2026-07-05T11:49:54Z"},{"alias_kind":"arxiv_version","alias_value":"2505.08827v2","created_at":"2026-07-05T11:49:54Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.08827","created_at":"2026-07-05T11:49:54Z"},{"alias_kind":"pith_short_12","alias_value":"HHB3ZSMZ4ENU","created_at":"2026-07-05T11:49:54Z"},{"alias_kind":"pith_short_16","alias_value":"HHB3ZSMZ4ENUFEU2","created_at":"2026-07-05T11:49:54Z"},{"alias_kind":"pith_short_8","alias_value":"HHB3ZSMZ","created_at":"2026-07-05T11:49:54Z"}],"graph_snapshots":[{"event_id":"sha256:39c1f47c43edb4bedbac0fc274594759b7eed2ba166091e1f47bffcc4c745ab4","target":"graph","created_at":"2026-07-05T11:49:54Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2505.08827/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Large language models can generate solutions to complex problems, but training them with reinforcement learning typically requires verifiable rewards that are expensive to create and not possible for all domains. We demonstrate that LLMs can effectively self-improve through self-judging without reference solutions, leveraging the inherent asymmetry between generating and verifying solutions. Our experiments show that models can provide reliable reward signals without ground truth answers, enabling reinforcement learning in domains where verifiable rewards are impractical. By implementing self-","authors_text":"Akira Yoshiyama, Dominique Garmier, Kevin Lopez, Toby Simonds","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-12T23:51:04Z","title":"RLSR: Reinforcement Learning from Self Reward"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.08827","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:cfe80d7750efc2b0d8e638fce18e8213e59a769ae6e642997d74505013197d12","target":"record","created_at":"2026-07-05T11:49:54Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"ddb1f091996ac3666088636d3724ec257bc706b0d638007c878b5ef9a932e396","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-12T23:51:04Z","title_canon_sha256":"8479c28cfa0a570a26088c614a6295f23a077919c8be4eb7b9fa19ea3c8384de"},"schema_version":"1.0","source":{"id":"2505.08827","kind":"arxiv","version":2}},"canonical_sha256":"39c3bcc999e11b42929a503aa412350a6be9e0fe1cf12d9e08f1eaadb252fd71","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"39c3bcc999e11b42929a503aa412350a6be9e0fe1cf12d9e08f1eaadb252fd71","first_computed_at":"2026-07-05T11:49:54.638488Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:49:54.638488Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"uV5YZDexhTn5SauNg65pGjy7JLdVocJwbRrfUiTMNo324xmC37Zp8y/gMzDqIBEPgzoBAcEq7mCjDPssw40EAQ==","signature_status":"signed_v1","signed_at":"2026-07-05T11:49:54.638934Z","signed_message":"canonical_sha256_bytes"},"source_id":"2505.08827","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:cfe80d7750efc2b0d8e638fce18e8213e59a769ae6e642997d74505013197d12","sha256:39c1f47c43edb4bedbac0fc274594759b7eed2ba166091e1f47bffcc4c745ab4"],"state_sha256":"66d7fd900b42dfe3474d683fb86cd73771579a31a1c6c5eccb2520b785224c96"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"vVUWHxMWNkbJQjQzeJ9yQcs6GGiaC6TWbJUIKP1boktsxTWn4yoOcPvqY8NnXPEgMK28W2BgVEAoR3KjH6doCw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-15T04:50:23.120175Z","bundle_sha256":"94900cabe718de445dfc46e3f7c65305e668e7299a0146268c009373d8f57384"}}