{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:QEU3NQFQAROYHPZFGDR4SFAKGJ","short_pith_number":"pith:QEU3NQFQ","schema_version":"1.0","canonical_sha256":"8129b6c0b0045d83bf2530e3c9140a324aaf5f25078f833790e2f8dbd7fe26c2","source":{"kind":"arxiv","id":"2502.08640","version":2},"attestation_state":"computed","paper":{"title":"Utility Engineering: Analyzing and Controlling Emergent Value Systems in AIs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV","cs.CY"],"primary_cat":"cs.LG","authors_text":"Adam Khoja, Bruce W. Lee, Dan Hendrycks, JaeHyuk Lim, Long Phan, Mantas Mazeika, Norman Mu, Oliver Zhang, Richard Ren, Rishub Tamirisa, Xuwang Yin","submitted_at":"2025-02-12T18:55:43Z","abstract_excerpt":"As AIs rapidly advance and become more agentic, the risk they pose is governed not only by their capabilities but increasingly by their propensities, including goals and values. Tracking the emergence of goals and values has proven a longstanding problem, and despite much interest over the years it remains unclear whether current AIs have meaningful values. We propose a solution to this problem, leveraging the framework of utility functions to study the internal coherence of AI preferences. Surprisingly, we find that independently-sampled preferences in current LLMs exhibit high degrees of str"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.08640","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-12T18:55:43Z","cross_cats_sorted":["cs.AI","cs.CL","cs.CV","cs.CY"],"title_canon_sha256":"a0e47a4fb080da47b63607585b8e18af2e0f73c8219a7e2642517e419b26ca16","abstract_canon_sha256":"3aad2873ca996cdb1dc099cdce746b1a496d8df3966f0e55165f04f3e7bf9441"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:16:41.782075Z","signature_b64":"ZJ/tjwdAipKKdTycYTzoiUH7XECu9Rg9wjkDqFmVjcSt3pzeoJPrTGkgV+BehHxEnFsZniOxDTyr2E549IJBDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8129b6c0b0045d83bf2530e3c9140a324aaf5f25078f833790e2f8dbd7fe26c2","last_reissued_at":"2026-07-05T10:16:41.781601Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:16:41.781601Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Utility Engineering: Analyzing and Controlling Emergent Value Systems in AIs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV","cs.CY"],"primary_cat":"cs.LG","authors_text":"Adam Khoja, Bruce W. Lee, Dan Hendrycks, JaeHyuk Lim, Long Phan, Mantas Mazeika, Norman Mu, Oliver Zhang, Richard Ren, Rishub Tamirisa, Xuwang Yin","submitted_at":"2025-02-12T18:55:43Z","abstract_excerpt":"As AIs rapidly advance and become more agentic, the risk they pose is governed not only by their capabilities but increasingly by their propensities, including goals and values. Tracking the emergence of goals and values has proven a longstanding problem, and despite much interest over the years it remains unclear whether current AIs have meaningful values. We propose a solution to this problem, leveraging the framework of utility functions to study the internal coherence of AI preferences. Surprisingly, we find that independently-sampled preferences in current LLMs exhibit high degrees of str"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.08640","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.08640/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.08640","created_at":"2026-07-05T10:16:41.781659+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.08640v2","created_at":"2026-07-05T10:16:41.781659+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.08640","created_at":"2026-07-05T10:16:41.781659+00:00"},{"alias_kind":"pith_short_12","alias_value":"QEU3NQFQAROY","created_at":"2026-07-05T10:16:41.781659+00:00"},{"alias_kind":"pith_short_16","alias_value":"QEU3NQFQAROYHPZF","created_at":"2026-07-05T10:16:41.781659+00:00"},{"alias_kind":"pith_short_8","alias_value":"QEU3NQFQ","created_at":"2026-07-05T10:16:41.781659+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08695","citing_title":"Artificial Persons","ref_index":4,"is_internal_anchor":true},{"citing_arxiv_id":"2607.02047","citing_title":"OpenSafeIntent: Evaluating Intent-Calibrated Safe Completion Across Dual-Use Prompt Sets","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22771","citing_title":"Reducing Political Manipulation with Consistency Training","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23565","citing_title":"Understanding Goal Generalisation in Sequential Reinforcement Learning","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2408.09049","citing_title":"Inertia in Moral and Value Judgments of Large Language Models","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22771","citing_title":"Reducing Political Manipulation with Consistency Training","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13339","citing_title":"Probing Persona-Dependent Preferences in Language Models","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16872","citing_title":"Some[Body] Must Receive That Pain for Agent Accountability","ref_index":92,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08556","citing_title":"Can Revealed Preferences Clarify LLM Alignment and Steering?","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17596","citing_title":"Terminal Wrench: A Dataset of 331 Reward-Hackable Environments and 3,632 Exploit Trajectories","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21864","citing_title":"FAccT-Checked: A Narrative Review of Authority Reconfigurations and Retention in AI-Mediated Journalism","ref_index":128,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QEU3NQFQAROYHPZFGDR4SFAKGJ","json":"https://pith.science/pith/QEU3NQFQAROYHPZFGDR4SFAKGJ.json","graph_json":"https://pith.science/api/pith-number/QEU3NQFQAROYHPZFGDR4SFAKGJ/graph.json","events_json":"https://pith.science/api/pith-number/QEU3NQFQAROYHPZFGDR4SFAKGJ/events.json","paper":"https://pith.science/paper/QEU3NQFQ"},"agent_actions":{"view_html":"https://pith.science/pith/QEU3NQFQAROYHPZFGDR4SFAKGJ","download_json":"https://pith.science/pith/QEU3NQFQAROYHPZFGDR4SFAKGJ.json","view_paper":"https://pith.science/paper/QEU3NQFQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.08640&json=true","fetch_graph":"https://pith.science/api/pith-number/QEU3NQFQAROYHPZFGDR4SFAKGJ/graph.json","fetch_events":"https://pith.science/api/pith-number/QEU3NQFQAROYHPZFGDR4SFAKGJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QEU3NQFQAROYHPZFGDR4SFAKGJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QEU3NQFQAROYHPZFGDR4SFAKGJ/action/storage_attestation","attest_author":"https://pith.science/pith/QEU3NQFQAROYHPZFGDR4SFAKGJ/action/author_attestation","sign_citation":"https://pith.science/pith/QEU3NQFQAROYHPZFGDR4SFAKGJ/action/citation_signature","submit_replication":"https://pith.science/pith/QEU3NQFQAROYHPZFGDR4SFAKGJ/action/replication_record"}},"created_at":"2026-07-05T10:16:41.781659+00:00","updated_at":"2026-07-05T10:16:41.781659+00:00"}