{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:TAPZT2LQKDPYU6772EVUUDQUBA","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"f71679c64b13faff4d32b5600172af2278240b581f70e675e1934ad4f99cd011","cross_cats_sorted":["cs.AI","cs.CL"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-03-15T21:43:46Z","title_canon_sha256":"1144e2e116f9fe11b9352c5cbdf0dfbb319446bcc381a67bbef6a344bacb86f7"},"schema_version":"1.0","source":{"id":"2403.10704","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2403.10704","created_at":"2026-07-05T09:06:25Z"},{"alias_kind":"arxiv_version","alias_value":"2403.10704v2","created_at":"2026-07-05T09:06:25Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.10704","created_at":"2026-07-05T09:06:25Z"},{"alias_kind":"pith_short_12","alias_value":"TAPZT2LQKDPY","created_at":"2026-07-05T09:06:25Z"},{"alias_kind":"pith_short_16","alias_value":"TAPZT2LQKDPYU677","created_at":"2026-07-05T09:06:25Z"},{"alias_kind":"pith_short_8","alias_value":"TAPZT2LQ","created_at":"2026-07-05T09:06:25Z"}],"graph_snapshots":[{"event_id":"sha256:b72b2d41fe2bdc010e141d5acedd653616e8e1fa4391e526187940b2802564df","target":"graph","created_at":"2026-07-05T09:06:25Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2403.10704/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"While Reinforcement Learning from Human Feedback (RLHF) effectively aligns pretrained Large Language and Vision-Language Models (LLMs, and VLMs) with human preferences, its computational cost and complexity hamper its wider adoption. To alleviate some of the computational burden of fine-tuning, parameter efficient methods, like LoRA were introduced. In this work, we empirically evaluate the setup of Parameter Efficient Reinforcement Learning from Human Feedback (PE-RLHF) that leverages LoRA fine-tuning for Reward Modeling, and Reinforcement Learning. We benchmark the PE-RLHF setup on six diver","authors_text":"Abhinav Rastogi, Alex Hutcheson, Bill Byrne, Bowen Li, Christiane Ahlheim, Hakim Sidahmed, Hassan Mansoor, Jarvis Jin, Jessica Hoffmann, Lucas Dixon, Roman Komarytsia, Samrat Phatale, Saravanan Ganesh, Simral Chaudhary, Wei Li, Yonghao Zhu, Zac Yu, Zhang Chen, Zhuonan Lin","cross_cats":["cs.AI","cs.CL"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-03-15T21:43:46Z","title":"Parameter Efficient Reinforcement Learning from Human Feedback"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.10704","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:f0611045d71f742229ecab17f9f8c1829cb929b4ec8ff1830100d048acf91260","target":"record","created_at":"2026-07-05T09:06:25Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"f71679c64b13faff4d32b5600172af2278240b581f70e675e1934ad4f99cd011","cross_cats_sorted":["cs.AI","cs.CL"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-03-15T21:43:46Z","title_canon_sha256":"1144e2e116f9fe11b9352c5cbdf0dfbb319446bcc381a67bbef6a344bacb86f7"},"schema_version":"1.0","source":{"id":"2403.10704","kind":"arxiv","version":2}},"canonical_sha256":"981f99e97050df8a7bffd12b4a0e140808439cfa93f0b2451e0b60b638dfe88e","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"981f99e97050df8a7bffd12b4a0e140808439cfa93f0b2451e0b60b638dfe88e","first_computed_at":"2026-07-05T09:06:25.555606Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T09:06:25.555606Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"fCrqlJtFhF8xPZmq4cd9JLgZu6WqG8ohO7WmKSIX9EIHLcl4FL3fWFSKNV2uNI09lMiy9UPtFzx2QBI1tAy+Bw==","signature_status":"signed_v1","signed_at":"2026-07-05T09:06:25.556142Z","signed_message":"canonical_sha256_bytes"},"source_id":"2403.10704","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:f0611045d71f742229ecab17f9f8c1829cb929b4ec8ff1830100d048acf91260","sha256:b72b2d41fe2bdc010e141d5acedd653616e8e1fa4391e526187940b2802564df"],"state_sha256":"a482e3a9c299465d3ce2b3c13fb8d3a0a1e97f4ee91b462ec3557c248302824d"}