{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2023:5VV4JTYNBCHWMPBCYBMUJLFSD6","short_pith_number":"pith:5VV4JTYN","canonical_record":{"source":{"id":"2310.13595","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CY","submitted_at":"2023-10-20T15:45:16Z","cross_cats_sorted":[],"title_canon_sha256":"633278139518b6cce0085a178f61d87c8cc7eef7dba2b8b6108baa8b8bec93bf","abstract_canon_sha256":"40714ee5528c3c574c03f02e4a120266d464dddf825ed7819d32dd0b387ef5c1"},"schema_version":"1.0"},"canonical_sha256":"ed6bc4cf0d088f663c22c05944acb21f975399d8b58db23ce51fa06e22e9f576","source":{"kind":"arxiv","id":"2310.13595","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2310.13595","created_at":"2026-07-05T07:17:33Z"},{"alias_kind":"arxiv_version","alias_value":"2310.13595v2","created_at":"2026-07-05T07:17:33Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.13595","created_at":"2026-07-05T07:17:33Z"},{"alias_kind":"pith_short_12","alias_value":"5VV4JTYNBCHW","created_at":"2026-07-05T07:17:33Z"},{"alias_kind":"pith_short_16","alias_value":"5VV4JTYNBCHWMPBC","created_at":"2026-07-05T07:17:33Z"},{"alias_kind":"pith_short_8","alias_value":"5VV4JTYN","created_at":"2026-07-05T07:17:33Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2023:5VV4JTYNBCHWMPBCYBMUJLFSD6","target":"record","payload":{"canonical_record":{"source":{"id":"2310.13595","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CY","submitted_at":"2023-10-20T15:45:16Z","cross_cats_sorted":[],"title_canon_sha256":"633278139518b6cce0085a178f61d87c8cc7eef7dba2b8b6108baa8b8bec93bf","abstract_canon_sha256":"40714ee5528c3c574c03f02e4a120266d464dddf825ed7819d32dd0b387ef5c1"},"schema_version":"1.0"},"canonical_sha256":"ed6bc4cf0d088f663c22c05944acb21f975399d8b58db23ce51fa06e22e9f576","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:17:33.303292Z","signature_b64":"WZWfM0KgbcA0Hpku7Nkui1YQBSGcAoCBU8LAkM6xsb96dqu6HSYtVATkayJ8e7xjXkxGctjWAyNclKWkpTEoAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ed6bc4cf0d088f663c22c05944acb21f975399d8b58db23ce51fa06e22e9f576","last_reissued_at":"2026-07-05T07:17:33.302800Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:17:33.302800Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2310.13595","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:17:33Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"SqTICbCSmV0NRSjoGNjQG7CmZkgnGD5DXjA7QRdwehvcKptk2iZf3MT2DzALU3sjrMt7/fdFW3I8Jj+cSKk8AQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-13T07:32:26.942931Z"},"content_sha256":"b760a031c18994f2a77dc1700477df613420a78bb4d9c73ef628de99081fb73e","schema_version":"1.0","event_id":"sha256:b760a031c18994f2a77dc1700477df613420a78bb4d9c73ef628de99081fb73e"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2023:5VV4JTYNBCHWMPBCYBMUJLFSD6","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"The History and Risks of Reinforcement Learning and Human Feedback","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CY","authors_text":"Nathan Lambert, Thomas Krendl Gilbert, Tom Zick","submitted_at":"2023-10-20T15:45:16Z","abstract_excerpt":"Reinforcement learning from human feedback (RLHF) has emerged as a powerful technique to make large language models (LLMs) easier to use and more effective. A core piece of the RLHF process is the training and utilization of a model of human preferences that acts as a reward function for optimization. This approach, which operates at the intersection of many stakeholders and academic disciplines, remains poorly understood. RLHF reward models are often cited as being central to achieving performance, yet very few descriptors of capabilities, evaluations, training methods, or open-source models "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.13595","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.13595/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:17:33Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"CknLOOEVZGDawzlBn3LUhUaEQH+srx2Sikuj/sd0QUHRJi6brn0pJUiIb3ZOHb+GETL/xz0ikom/IXNZW3UtDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-13T07:32:26.943438Z"},"content_sha256":"cb3396f7af9cc34e6d8b150f867325d5fe6a51a992456bf707381d7588c0855c","schema_version":"1.0","event_id":"sha256:cb3396f7af9cc34e6d8b150f867325d5fe6a51a992456bf707381d7588c0855c"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/5VV4JTYNBCHWMPBCYBMUJLFSD6/bundle.json","state_url":"https://pith.science/pith/5VV4JTYNBCHWMPBCYBMUJLFSD6/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/5VV4JTYNBCHWMPBCYBMUJLFSD6/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-13T07:32:26Z","links":{"resolver":"https://pith.science/pith/5VV4JTYNBCHWMPBCYBMUJLFSD6","bundle":"https://pith.science/pith/5VV4JTYNBCHWMPBCYBMUJLFSD6/bundle.json","state":"https://pith.science/pith/5VV4JTYNBCHWMPBCYBMUJLFSD6/state.json","well_known_bundle":"https://pith.science/.well-known/pith/5VV4JTYNBCHWMPBCYBMUJLFSD6/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:5VV4JTYNBCHWMPBCYBMUJLFSD6","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"40714ee5528c3c574c03f02e4a120266d464dddf825ed7819d32dd0b387ef5c1","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CY","submitted_at":"2023-10-20T15:45:16Z","title_canon_sha256":"633278139518b6cce0085a178f61d87c8cc7eef7dba2b8b6108baa8b8bec93bf"},"schema_version":"1.0","source":{"id":"2310.13595","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2310.13595","created_at":"2026-07-05T07:17:33Z"},{"alias_kind":"arxiv_version","alias_value":"2310.13595v2","created_at":"2026-07-05T07:17:33Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.13595","created_at":"2026-07-05T07:17:33Z"},{"alias_kind":"pith_short_12","alias_value":"5VV4JTYNBCHW","created_at":"2026-07-05T07:17:33Z"},{"alias_kind":"pith_short_16","alias_value":"5VV4JTYNBCHWMPBC","created_at":"2026-07-05T07:17:33Z"},{"alias_kind":"pith_short_8","alias_value":"5VV4JTYN","created_at":"2026-07-05T07:17:33Z"}],"graph_snapshots":[{"event_id":"sha256:cb3396f7af9cc34e6d8b150f867325d5fe6a51a992456bf707381d7588c0855c","target":"graph","created_at":"2026-07-05T07:17:33Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2310.13595/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning from human feedback (RLHF) has emerged as a powerful technique to make large language models (LLMs) easier to use and more effective. A core piece of the RLHF process is the training and utilization of a model of human preferences that acts as a reward function for optimization. This approach, which operates at the intersection of many stakeholders and academic disciplines, remains poorly understood. RLHF reward models are often cited as being central to achieving performance, yet very few descriptors of capabilities, evaluations, training methods, or open-source models ","authors_text":"Nathan Lambert, Thomas Krendl Gilbert, Tom Zick","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CY","submitted_at":"2023-10-20T15:45:16Z","title":"The History and Risks of Reinforcement Learning and Human Feedback"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.13595","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:b760a031c18994f2a77dc1700477df613420a78bb4d9c73ef628de99081fb73e","target":"record","created_at":"2026-07-05T07:17:33Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"40714ee5528c3c574c03f02e4a120266d464dddf825ed7819d32dd0b387ef5c1","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CY","submitted_at":"2023-10-20T15:45:16Z","title_canon_sha256":"633278139518b6cce0085a178f61d87c8cc7eef7dba2b8b6108baa8b8bec93bf"},"schema_version":"1.0","source":{"id":"2310.13595","kind":"arxiv","version":2}},"canonical_sha256":"ed6bc4cf0d088f663c22c05944acb21f975399d8b58db23ce51fa06e22e9f576","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"ed6bc4cf0d088f663c22c05944acb21f975399d8b58db23ce51fa06e22e9f576","first_computed_at":"2026-07-05T07:17:33.302800Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T07:17:33.302800Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"WZWfM0KgbcA0Hpku7Nkui1YQBSGcAoCBU8LAkM6xsb96dqu6HSYtVATkayJ8e7xjXkxGctjWAyNclKWkpTEoAQ==","signature_status":"signed_v1","signed_at":"2026-07-05T07:17:33.303292Z","signed_message":"canonical_sha256_bytes"},"source_id":"2310.13595","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:b760a031c18994f2a77dc1700477df613420a78bb4d9c73ef628de99081fb73e","sha256:cb3396f7af9cc34e6d8b150f867325d5fe6a51a992456bf707381d7588c0855c"],"state_sha256":"fdf500371fdb7a959edc7a16d62565563a50b5837d5d7bccd1fb0b044a9a6b39"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"+1jFA/zEahGquLXBiW8G/LGs7il9J973xi8BM1W0mL4phvoEsZXow4/0MVMcTown0XlSlR84tiYSUU3XDZBACg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-13T07:32:26.949647Z","bundle_sha256":"2ea5a5a2cff5ece9516dc511662797e44da7d584e2c00847abe8208a1e927cc3"}}