{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:SC5JP674CIUXVPVGXGW24XMH6L","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"4628beaa4e2aa1acb53750d94d8d4089713a54c2311cb9e189e15cd6b86298c4","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-06-21T18:57:38Z","title_canon_sha256":"48d1a26caf8152e778f210930d216ea8a54f801b5b83d28c05ab4303d4f95d5c"},"schema_version":"1.0","source":{"id":"2406.15599","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2406.15599","created_at":"2026-07-05T10:10:41Z"},{"alias_kind":"arxiv_version","alias_value":"2406.15599v2","created_at":"2026-07-05T10:10:41Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.15599","created_at":"2026-07-05T10:10:41Z"},{"alias_kind":"pith_short_12","alias_value":"SC5JP674CIUX","created_at":"2026-07-05T10:10:41Z"},{"alias_kind":"pith_short_16","alias_value":"SC5JP674CIUXVPVG","created_at":"2026-07-05T10:10:41Z"},{"alias_kind":"pith_short_8","alias_value":"SC5JP674","created_at":"2026-07-05T10:10:41Z"}],"graph_snapshots":[{"event_id":"sha256:dbdb64d5d7a4477fe35ea30293251eccb14efb04f26aa2adbd1e4ce76f056eee","target":"graph","created_at":"2026-07-05T10:10:41Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2406.15599/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Ensuring AI models align with human values is essential for their safety and functionality. Reinforcement learning from human feedback (RLHF) leverages human preferences to achieve this alignment. However, when preferences are sourced from diverse populations, point estimates of reward can result in suboptimal performance or be unfair to specific groups. We propose Pareto Optimal Preference Learning (POPL), which enables pluralistic alignment by framing discrepant group preferences as objectives with potential trade-offs, aiming for policies that are Pareto-optimal on the preference dataset. P","authors_text":"Lee Spector, Li Ding, Ryan Bahlous-Boldi, Scott Niekum","cross_cats":["cs.AI"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-06-21T18:57:38Z","title":"Pareto-Optimal Learning from Preferences with Hidden Context"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.15599","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:c34974c4b6319f7e52a0389c2a50f6769ba4e045673a917ede7ce137bee3f20f","target":"record","created_at":"2026-07-05T10:10:41Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"4628beaa4e2aa1acb53750d94d8d4089713a54c2311cb9e189e15cd6b86298c4","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-06-21T18:57:38Z","title_canon_sha256":"48d1a26caf8152e778f210930d216ea8a54f801b5b83d28c05ab4303d4f95d5c"},"schema_version":"1.0","source":{"id":"2406.15599","kind":"arxiv","version":2}},"canonical_sha256":"90ba97fbfc12297abea6b9adae5d87f2ecde644e3edff7d32000398c83f19916","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"90ba97fbfc12297abea6b9adae5d87f2ecde644e3edff7d32000398c83f19916","first_computed_at":"2026-07-05T10:10:41.406854Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T10:10:41.406854Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"Qeyl5G54NWT9phWUGIglKgu8edeb0huQHC7dSwwFC6SEo8Obk1QzFNhmCJbhmo/RABtHSFGCNf6yWQRjalWqAA==","signature_status":"signed_v1","signed_at":"2026-07-05T10:10:41.407271Z","signed_message":"canonical_sha256_bytes"},"source_id":"2406.15599","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:c34974c4b6319f7e52a0389c2a50f6769ba4e045673a917ede7ce137bee3f20f","sha256:dbdb64d5d7a4477fe35ea30293251eccb14efb04f26aa2adbd1e4ce76f056eee"],"state_sha256":"b4dc32bb81aef38a12bd5bd33a85b28ed529732d9bba46a62bbe7c07f822b4f0"}