{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:HHJ3RGZZG4MB556ACCY3AZWKMJ","short_pith_number":"pith:HHJ3RGZZ","schema_version":"1.0","canonical_sha256":"39d3b89b3937181ef7c010b1b066ca6245f78e79df0731b0700a52a33e91b445","source":{"kind":"arxiv","id":"2606.20624","version":1},"attestation_state":"computed","paper":{"title":"In LLM Reasoning, there is Irrationality on top of Value Misalignment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG","stat.ML"],"primary_cat":"cs.AI","authors_text":"Fengxiang He, Kejiang Qian","submitted_at":"2026-05-26T14:26:11Z","abstract_excerpt":"Significant progress has been made in aligning LLMs with target value functions. We argue that, even when an LLM has been well aligned in (post-)training, it may still fail to maximise the aligned value in reasoning. We mathematically formalise this gap as rational value risk: the utility discrepancy between a model's deployed reasoning strategy and its rational counterpart, which is defined to be the responses that maximise expected utility in the steepest direction. The estimation error of rational value risk is further decomposed into three components from finite candidates, finite prompts,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2606.20624","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-05-26T14:26:11Z","cross_cats_sorted":["cs.CL","cs.LG","stat.ML"],"title_canon_sha256":"042ba9a904f21382dc32a3b82e8eea29455970e1f394eac7f490b02837bdbca7","abstract_canon_sha256":"1abfdd655a930cf10498ea7cee4c889e5a5084b4f0cf79ab9b7b7e8ef4acd184"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-06-23T00:11:51.712247Z","signature_b64":"b2iW77Izvyoj2xd4RV/1e01fQXoA8q+dZaJAm0fsuGjyjmpmVDjh5+GxixnAfmBwJbf9weV8aOElIRRzU9BiAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"39d3b89b3937181ef7c010b1b066ca6245f78e79df0731b0700a52a33e91b445","last_reissued_at":"2026-06-23T00:11:51.711848Z","signature_status":"signed_v1","first_computed_at":"2026-06-23T00:11:51.711848Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"In LLM Reasoning, there is Irrationality on top of Value Misalignment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG","stat.ML"],"primary_cat":"cs.AI","authors_text":"Fengxiang He, Kejiang Qian","submitted_at":"2026-05-26T14:26:11Z","abstract_excerpt":"Significant progress has been made in aligning LLMs with target value functions. We argue that, even when an LLM has been well aligned in (post-)training, it may still fail to maximise the aligned value in reasoning. We mathematically formalise this gap as rational value risk: the utility discrepancy between a model's deployed reasoning strategy and its rational counterpart, which is defined to be the responses that maximise expected utility in the steepest direction. The estimation error of rational value risk is further decomposed into three components from finite candidates, finite prompts,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2606.20624","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2606.20624/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2606.20624","created_at":"2026-06-23T00:11:51.711906+00:00"},{"alias_kind":"arxiv_version","alias_value":"2606.20624v1","created_at":"2026-06-23T00:11:51.711906+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2606.20624","created_at":"2026-06-23T00:11:51.711906+00:00"},{"alias_kind":"pith_short_12","alias_value":"HHJ3RGZZG4MB","created_at":"2026-06-23T00:11:51.711906+00:00"},{"alias_kind":"pith_short_16","alias_value":"HHJ3RGZZG4MB556A","created_at":"2026-06-23T00:11:51.711906+00:00"},{"alias_kind":"pith_short_8","alias_value":"HHJ3RGZZ","created_at":"2026-06-23T00:11:51.711906+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HHJ3RGZZG4MB556ACCY3AZWKMJ","json":"https://pith.science/pith/HHJ3RGZZG4MB556ACCY3AZWKMJ.json","graph_json":"https://pith.science/api/pith-number/HHJ3RGZZG4MB556ACCY3AZWKMJ/graph.json","events_json":"https://pith.science/api/pith-number/HHJ3RGZZG4MB556ACCY3AZWKMJ/events.json","paper":"https://pith.science/paper/HHJ3RGZZ"},"agent_actions":{"view_html":"https://pith.science/pith/HHJ3RGZZG4MB556ACCY3AZWKMJ","download_json":"https://pith.science/pith/HHJ3RGZZG4MB556ACCY3AZWKMJ.json","view_paper":"https://pith.science/paper/HHJ3RGZZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2606.20624&json=true","fetch_graph":"https://pith.science/api/pith-number/HHJ3RGZZG4MB556ACCY3AZWKMJ/graph.json","fetch_events":"https://pith.science/api/pith-number/HHJ3RGZZG4MB556ACCY3AZWKMJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HHJ3RGZZG4MB556ACCY3AZWKMJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HHJ3RGZZG4MB556ACCY3AZWKMJ/action/storage_attestation","attest_author":"https://pith.science/pith/HHJ3RGZZG4MB556ACCY3AZWKMJ/action/author_attestation","sign_citation":"https://pith.science/pith/HHJ3RGZZG4MB556ACCY3AZWKMJ/action/citation_signature","submit_replication":"https://pith.science/pith/HHJ3RGZZG4MB556ACCY3AZWKMJ/action/replication_record"}},"created_at":"2026-06-23T00:11:51.711906+00:00","updated_at":"2026-06-23T00:11:51.711906+00:00"}