{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TAPZT2LQKDPYU6772EVUUDQUBA","short_pith_number":"pith:TAPZT2LQ","schema_version":"1.0","canonical_sha256":"981f99e97050df8a7bffd12b4a0e140808439cfa93f0b2451e0b60b638dfe88e","source":{"kind":"arxiv","id":"2403.10704","version":2},"attestation_state":"computed","paper":{"title":"Parameter Efficient Reinforcement Learning from Human Feedback","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Abhinav Rastogi, Alex Hutcheson, Bill Byrne, Bowen Li, Christiane Ahlheim, Hakim Sidahmed, Hassan Mansoor, Jarvis Jin, Jessica Hoffmann, Lucas Dixon, Roman Komarytsia, Samrat Phatale, Saravanan Ganesh, Simral Chaudhary, Wei Li, Yonghao Zhu, Zac Yu, Zhang Chen, Zhuonan Lin","submitted_at":"2024-03-15T21:43:46Z","abstract_excerpt":"While Reinforcement Learning from Human Feedback (RLHF) effectively aligns pretrained Large Language and Vision-Language Models (LLMs, and VLMs) with human preferences, its computational cost and complexity hamper its wider adoption. To alleviate some of the computational burden of fine-tuning, parameter efficient methods, like LoRA were introduced. In this work, we empirically evaluate the setup of Parameter Efficient Reinforcement Learning from Human Feedback (PE-RLHF) that leverages LoRA fine-tuning for Reward Modeling, and Reinforcement Learning. We benchmark the PE-RLHF setup on six diver"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.10704","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-03-15T21:43:46Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"1144e2e116f9fe11b9352c5cbdf0dfbb319446bcc381a67bbef6a344bacb86f7","abstract_canon_sha256":"f71679c64b13faff4d32b5600172af2278240b581f70e675e1934ad4f99cd011"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:06:25.556142Z","signature_b64":"fCrqlJtFhF8xPZmq4cd9JLgZu6WqG8ohO7WmKSIX9EIHLcl4FL3fWFSKNV2uNI09lMiy9UPtFzx2QBI1tAy+Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"981f99e97050df8a7bffd12b4a0e140808439cfa93f0b2451e0b60b638dfe88e","last_reissued_at":"2026-07-05T09:06:25.555606Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:06:25.555606Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Parameter Efficient Reinforcement Learning from Human Feedback","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Abhinav Rastogi, Alex Hutcheson, Bill Byrne, Bowen Li, Christiane Ahlheim, Hakim Sidahmed, Hassan Mansoor, Jarvis Jin, Jessica Hoffmann, Lucas Dixon, Roman Komarytsia, Samrat Phatale, Saravanan Ganesh, Simral Chaudhary, Wei Li, Yonghao Zhu, Zac Yu, Zhang Chen, Zhuonan Lin","submitted_at":"2024-03-15T21:43:46Z","abstract_excerpt":"While Reinforcement Learning from Human Feedback (RLHF) effectively aligns pretrained Large Language and Vision-Language Models (LLMs, and VLMs) with human preferences, its computational cost and complexity hamper its wider adoption. To alleviate some of the computational burden of fine-tuning, parameter efficient methods, like LoRA were introduced. In this work, we empirically evaluate the setup of Parameter Efficient Reinforcement Learning from Human Feedback (PE-RLHF) that leverages LoRA fine-tuning for Reward Modeling, and Reinforcement Learning. We benchmark the PE-RLHF setup on six diver"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.10704","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.10704/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.10704","created_at":"2026-07-05T09:06:25.555675+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.10704v2","created_at":"2026-07-05T09:06:25.555675+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.10704","created_at":"2026-07-05T09:06:25.555675+00:00"},{"alias_kind":"pith_short_12","alias_value":"TAPZT2LQKDPY","created_at":"2026-07-05T09:06:25.555675+00:00"},{"alias_kind":"pith_short_16","alias_value":"TAPZT2LQKDPYU677","created_at":"2026-07-05T09:06:25.555675+00:00"},{"alias_kind":"pith_short_8","alias_value":"TAPZT2LQ","created_at":"2026-07-05T09:06:25.555675+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28401","citing_title":"Vision-driven Preference Synthesis for Mitigating Hallucinations in VLMs","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01123","citing_title":"PERSA: Reinforcement Learning for Professor-Style Personalized Feedback with LLMs","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16543","citing_title":"Conjunctive Prompt Attacks in Multi-Agent LLM Systems","ref_index":33,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TAPZT2LQKDPYU6772EVUUDQUBA","json":"https://pith.science/pith/TAPZT2LQKDPYU6772EVUUDQUBA.json","graph_json":"https://pith.science/api/pith-number/TAPZT2LQKDPYU6772EVUUDQUBA/graph.json","events_json":"https://pith.science/api/pith-number/TAPZT2LQKDPYU6772EVUUDQUBA/events.json","paper":"https://pith.science/paper/TAPZT2LQ"},"agent_actions":{"view_html":"https://pith.science/pith/TAPZT2LQKDPYU6772EVUUDQUBA","download_json":"https://pith.science/pith/TAPZT2LQKDPYU6772EVUUDQUBA.json","view_paper":"https://pith.science/paper/TAPZT2LQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.10704&json=true","fetch_graph":"https://pith.science/api/pith-number/TAPZT2LQKDPYU6772EVUUDQUBA/graph.json","fetch_events":"https://pith.science/api/pith-number/TAPZT2LQKDPYU6772EVUUDQUBA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TAPZT2LQKDPYU6772EVUUDQUBA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TAPZT2LQKDPYU6772EVUUDQUBA/action/storage_attestation","attest_author":"https://pith.science/pith/TAPZT2LQKDPYU6772EVUUDQUBA/action/author_attestation","sign_citation":"https://pith.science/pith/TAPZT2LQKDPYU6772EVUUDQUBA/action/citation_signature","submit_replication":"https://pith.science/pith/TAPZT2LQKDPYU6772EVUUDQUBA/action/replication_record"}},"created_at":"2026-07-05T09:06:25.555675+00:00","updated_at":"2026-07-05T09:06:25.555675+00:00"}