{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:RIUB5RACBJQL7EGEVUVT5YPBII","short_pith_number":"pith:RIUB5RAC","schema_version":"1.0","canonical_sha256":"8a281ec4020a60bf90c4ad2b3ee1e14237a643a9b70bcdc0e52a33af97e99329","source":{"kind":"arxiv","id":"2401.00243","version":1},"attestation_state":"computed","paper":{"title":"Uncertainty-Penalized Reinforcement Learning from Human Feedback with Diverse Reward LoRA Ensembles","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bo Ding, Dawei Feng, Han Zhang, Huaimin Wang, Kele Xu, Yuanzhao Zhai, Yue Yu, Yu Lei","submitted_at":"2023-12-30T14:14:14Z","abstract_excerpt":"Reinforcement learning from human feedback (RLHF) emerges as a promising paradigm for aligning large language models (LLMs). However, a notable challenge in RLHF is overoptimization, where beyond a certain threshold, the pursuit of higher rewards leads to a decline in human preferences. In this paper, we observe the weakness of KL regularization which is commonly employed in existing RLHF methods to address overoptimization. To mitigate this limitation, we scrutinize the RLHF objective in the offline dataset and propose uncertainty-penalized RLHF (UP-RLHF), which incorporates uncertainty regul"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.00243","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-12-30T14:14:14Z","cross_cats_sorted":[],"title_canon_sha256":"44baa034d026023c1e16ebb715568d6d2f8f0eda6145e1b5bb757b7fbc9664ba","abstract_canon_sha256":"5119e51677a320cff9580c51b4f4191f723589b0d67b6f4286902afb3ae964e9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:29:02.710299Z","signature_b64":"3S7zgy+zT/urUdEvh0y6Ems2EoEV600xtqTvbroh1O6FDkyIWgPR4uWBVr1EfKqsJbigHMPU79j1MVC1uB7+CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8a281ec4020a60bf90c4ad2b3ee1e14237a643a9b70bcdc0e52a33af97e99329","last_reissued_at":"2026-07-05T07:29:02.709690Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:29:02.709690Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Uncertainty-Penalized Reinforcement Learning from Human Feedback with Diverse Reward LoRA Ensembles","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bo Ding, Dawei Feng, Han Zhang, Huaimin Wang, Kele Xu, Yuanzhao Zhai, Yue Yu, Yu Lei","submitted_at":"2023-12-30T14:14:14Z","abstract_excerpt":"Reinforcement learning from human feedback (RLHF) emerges as a promising paradigm for aligning large language models (LLMs). However, a notable challenge in RLHF is overoptimization, where beyond a certain threshold, the pursuit of higher rewards leads to a decline in human preferences. In this paper, we observe the weakness of KL regularization which is commonly employed in existing RLHF methods to address overoptimization. To mitigate this limitation, we scrutinize the RLHF objective in the offline dataset and propose uncertainty-penalized RLHF (UP-RLHF), which incorporates uncertainty regul"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.00243","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.00243/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.00243","created_at":"2026-07-05T07:29:02.709772+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.00243v1","created_at":"2026-07-05T07:29:02.709772+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.00243","created_at":"2026-07-05T07:29:02.709772+00:00"},{"alias_kind":"pith_short_12","alias_value":"RIUB5RACBJQL","created_at":"2026-07-05T07:29:02.709772+00:00"},{"alias_kind":"pith_short_16","alias_value":"RIUB5RACBJQL7EGE","created_at":"2026-07-05T07:29:02.709772+00:00"},{"alias_kind":"pith_short_8","alias_value":"RIUB5RAC","created_at":"2026-07-05T07:29:02.709772+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09711","citing_title":"Proxy Reward Internalization and Mechanistic Exploitation: A Learned Precursor to Reward Hacking and Its Generalization","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09073","citing_title":"A Unifying Lens on Reward Uncertainty in RLHF","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22402","citing_title":"Reinforcement learning to improve large language model-based automated code compliance systems","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2410.06431","citing_title":"Functional-level Uncertainty Quantification for Calibrated Fine-tuning on LLMs","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00155","citing_title":"Wasserstein Distributionally Robust Regret Optimization for Reinforcement Learning from Human Feedback","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RIUB5RACBJQL7EGEVUVT5YPBII","json":"https://pith.science/pith/RIUB5RACBJQL7EGEVUVT5YPBII.json","graph_json":"https://pith.science/api/pith-number/RIUB5RACBJQL7EGEVUVT5YPBII/graph.json","events_json":"https://pith.science/api/pith-number/RIUB5RACBJQL7EGEVUVT5YPBII/events.json","paper":"https://pith.science/paper/RIUB5RAC"},"agent_actions":{"view_html":"https://pith.science/pith/RIUB5RACBJQL7EGEVUVT5YPBII","download_json":"https://pith.science/pith/RIUB5RACBJQL7EGEVUVT5YPBII.json","view_paper":"https://pith.science/paper/RIUB5RAC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.00243&json=true","fetch_graph":"https://pith.science/api/pith-number/RIUB5RACBJQL7EGEVUVT5YPBII/graph.json","fetch_events":"https://pith.science/api/pith-number/RIUB5RACBJQL7EGEVUVT5YPBII/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RIUB5RACBJQL7EGEVUVT5YPBII/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RIUB5RACBJQL7EGEVUVT5YPBII/action/storage_attestation","attest_author":"https://pith.science/pith/RIUB5RACBJQL7EGEVUVT5YPBII/action/author_attestation","sign_citation":"https://pith.science/pith/RIUB5RACBJQL7EGEVUVT5YPBII/action/citation_signature","submit_replication":"https://pith.science/pith/RIUB5RACBJQL7EGEVUVT5YPBII/action/replication_record"}},"created_at":"2026-07-05T07:29:02.709772+00:00","updated_at":"2026-07-05T07:29:02.709772+00:00"}