{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:SC5JP674CIUXVPVGXGW24XMH6L","short_pith_number":"pith:SC5JP674","schema_version":"1.0","canonical_sha256":"90ba97fbfc12297abea6b9adae5d87f2ecde644e3edff7d32000398c83f19916","source":{"kind":"arxiv","id":"2406.15599","version":2},"attestation_state":"computed","paper":{"title":"Pareto-Optimal Learning from Preferences with Hidden Context","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Lee Spector, Li Ding, Ryan Bahlous-Boldi, Scott Niekum","submitted_at":"2024-06-21T18:57:38Z","abstract_excerpt":"Ensuring AI models align with human values is essential for their safety and functionality. Reinforcement learning from human feedback (RLHF) leverages human preferences to achieve this alignment. However, when preferences are sourced from diverse populations, point estimates of reward can result in suboptimal performance or be unfair to specific groups. We propose Pareto Optimal Preference Learning (POPL), which enables pluralistic alignment by framing discrepant group preferences as objectives with potential trade-offs, aiming for policies that are Pareto-optimal on the preference dataset. P"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.15599","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-06-21T18:57:38Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"48d1a26caf8152e778f210930d216ea8a54f801b5b83d28c05ab4303d4f95d5c","abstract_canon_sha256":"4628beaa4e2aa1acb53750d94d8d4089713a54c2311cb9e189e15cd6b86298c4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:10:41.407271Z","signature_b64":"Qeyl5G54NWT9phWUGIglKgu8edeb0huQHC7dSwwFC6SEo8Obk1QzFNhmCJbhmo/RABtHSFGCNf6yWQRjalWqAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"90ba97fbfc12297abea6b9adae5d87f2ecde644e3edff7d32000398c83f19916","last_reissued_at":"2026-07-05T10:10:41.406854Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:10:41.406854Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Pareto-Optimal Learning from Preferences with Hidden Context","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Lee Spector, Li Ding, Ryan Bahlous-Boldi, Scott Niekum","submitted_at":"2024-06-21T18:57:38Z","abstract_excerpt":"Ensuring AI models align with human values is essential for their safety and functionality. Reinforcement learning from human feedback (RLHF) leverages human preferences to achieve this alignment. However, when preferences are sourced from diverse populations, point estimates of reward can result in suboptimal performance or be unfair to specific groups. We propose Pareto Optimal Preference Learning (POPL), which enables pluralistic alignment by framing discrepant group preferences as objectives with potential trade-offs, aiming for policies that are Pareto-optimal on the preference dataset. P"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.15599","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.15599/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.15599","created_at":"2026-07-05T10:10:41.406909+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.15599v2","created_at":"2026-07-05T10:10:41.406909+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.15599","created_at":"2026-07-05T10:10:41.406909+00:00"},{"alias_kind":"pith_short_12","alias_value":"SC5JP674CIUX","created_at":"2026-07-05T10:10:41.406909+00:00"},{"alias_kind":"pith_short_16","alias_value":"SC5JP674CIUXVPVG","created_at":"2026-07-05T10:10:41.406909+00:00"},{"alias_kind":"pith_short_8","alias_value":"SC5JP674","created_at":"2026-07-05T10:10:41.406909+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.13998","citing_title":"Few-shot Steerable Alignment: Adapting Rewards and LLM Policies with Neural Processes","ref_index":4,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SC5JP674CIUXVPVGXGW24XMH6L","json":"https://pith.science/pith/SC5JP674CIUXVPVGXGW24XMH6L.json","graph_json":"https://pith.science/api/pith-number/SC5JP674CIUXVPVGXGW24XMH6L/graph.json","events_json":"https://pith.science/api/pith-number/SC5JP674CIUXVPVGXGW24XMH6L/events.json","paper":"https://pith.science/paper/SC5JP674"},"agent_actions":{"view_html":"https://pith.science/pith/SC5JP674CIUXVPVGXGW24XMH6L","download_json":"https://pith.science/pith/SC5JP674CIUXVPVGXGW24XMH6L.json","view_paper":"https://pith.science/paper/SC5JP674","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.15599&json=true","fetch_graph":"https://pith.science/api/pith-number/SC5JP674CIUXVPVGXGW24XMH6L/graph.json","fetch_events":"https://pith.science/api/pith-number/SC5JP674CIUXVPVGXGW24XMH6L/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SC5JP674CIUXVPVGXGW24XMH6L/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SC5JP674CIUXVPVGXGW24XMH6L/action/storage_attestation","attest_author":"https://pith.science/pith/SC5JP674CIUXVPVGXGW24XMH6L/action/author_attestation","sign_citation":"https://pith.science/pith/SC5JP674CIUXVPVGXGW24XMH6L/action/citation_signature","submit_replication":"https://pith.science/pith/SC5JP674CIUXVPVGXGW24XMH6L/action/replication_record"}},"created_at":"2026-07-05T10:10:41.406909+00:00","updated_at":"2026-07-05T10:10:41.406909+00:00"}