{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:5JMHJUO2QIP6YH45ZNQQXHRFED","short_pith_number":"pith:5JMHJUO2","schema_version":"1.0","canonical_sha256":"ea5874d1da821fec1f9dcb610b9e2520dc02bf05612be412ae65b6fe727b5a6d","source":{"kind":"arxiv","id":"2012.01557","version":2},"attestation_state":"computed","paper":{"title":"Value Alignment Verification","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Anca D. Dragan, Daniel S. Brown, Jordan Schneider, Scott Niekum","submitted_at":"2020-12-02T22:04:01Z","abstract_excerpt":"As humans interact with autonomous agents to perform increasingly complicated, potentially risky tasks, it is important to be able to efficiently evaluate an agent's performance and correctness. In this paper we formalize and theoretically analyze the problem of efficient value alignment verification: how to efficiently test whether the behavior of another agent is aligned with a human's values. The goal is to construct a kind of \"driver's test\" that a human can give to any agent which will verify value alignment via a minimal number of queries. We study alignment verification problems with bo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2012.01557","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-12-02T22:04:01Z","cross_cats_sorted":[],"title_canon_sha256":"3b52fe37dba4185c1fca76e7aa47814653808060db97c97f71aad50fcbd21399","abstract_canon_sha256":"29e3d3717809fa3b21198a88866686452f07afe14bd916b8d6aff2bb9487c25e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:48:20.077038Z","signature_b64":"mcgOkS8LkC0d7rP3NRJ+dSPTglHVP4KSbxBJ/71pnUwyTPs1lMi/6Et9ZdYiE5GytUnlOZd98ZRAJid8IWeSAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ea5874d1da821fec1f9dcb610b9e2520dc02bf05612be412ae65b6fe727b5a6d","last_reissued_at":"2026-07-05T02:48:20.076580Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:48:20.076580Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Value Alignment Verification","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Anca D. Dragan, Daniel S. Brown, Jordan Schneider, Scott Niekum","submitted_at":"2020-12-02T22:04:01Z","abstract_excerpt":"As humans interact with autonomous agents to perform increasingly complicated, potentially risky tasks, it is important to be able to efficiently evaluate an agent's performance and correctness. In this paper we formalize and theoretically analyze the problem of efficient value alignment verification: how to efficiently test whether the behavior of another agent is aligned with a human's values. The goal is to construct a kind of \"driver's test\" that a human can give to any agent which will verify value alignment via a minimal number of queries. We study alignment verification problems with bo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2012.01557","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2012.01557/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2012.01557","created_at":"2026-07-05T02:48:20.076633+00:00"},{"alias_kind":"arxiv_version","alias_value":"2012.01557v2","created_at":"2026-07-05T02:48:20.076633+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2012.01557","created_at":"2026-07-05T02:48:20.076633+00:00"},{"alias_kind":"pith_short_12","alias_value":"5JMHJUO2QIP6","created_at":"2026-07-05T02:48:20.076633+00:00"},{"alias_kind":"pith_short_16","alias_value":"5JMHJUO2QIP6YH45","created_at":"2026-07-05T02:48:20.076633+00:00"},{"alias_kind":"pith_short_8","alias_value":"5JMHJUO2","created_at":"2026-07-05T02:48:20.076633+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2504.15088","citing_title":"Safety Co-Option and Compromised National Security: The Self-Fulfilling Prophecy of Weakened AI Risk Thresholds","ref_index":17,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5JMHJUO2QIP6YH45ZNQQXHRFED","json":"https://pith.science/pith/5JMHJUO2QIP6YH45ZNQQXHRFED.json","graph_json":"https://pith.science/api/pith-number/5JMHJUO2QIP6YH45ZNQQXHRFED/graph.json","events_json":"https://pith.science/api/pith-number/5JMHJUO2QIP6YH45ZNQQXHRFED/events.json","paper":"https://pith.science/paper/5JMHJUO2"},"agent_actions":{"view_html":"https://pith.science/pith/5JMHJUO2QIP6YH45ZNQQXHRFED","download_json":"https://pith.science/pith/5JMHJUO2QIP6YH45ZNQQXHRFED.json","view_paper":"https://pith.science/paper/5JMHJUO2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2012.01557&json=true","fetch_graph":"https://pith.science/api/pith-number/5JMHJUO2QIP6YH45ZNQQXHRFED/graph.json","fetch_events":"https://pith.science/api/pith-number/5JMHJUO2QIP6YH45ZNQQXHRFED/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5JMHJUO2QIP6YH45ZNQQXHRFED/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5JMHJUO2QIP6YH45ZNQQXHRFED/action/storage_attestation","attest_author":"https://pith.science/pith/5JMHJUO2QIP6YH45ZNQQXHRFED/action/author_attestation","sign_citation":"https://pith.science/pith/5JMHJUO2QIP6YH45ZNQQXHRFED/action/citation_signature","submit_replication":"https://pith.science/pith/5JMHJUO2QIP6YH45ZNQQXHRFED/action/replication_record"}},"created_at":"2026-07-05T02:48:20.076633+00:00","updated_at":"2026-07-05T02:48:20.076633+00:00"}