{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:UIAROYEXI6VNO6K7L34BG2IKPJ","short_pith_number":"pith:UIAROYEX","schema_version":"1.0","canonical_sha256":"a20117609747aad7795f5ef813690a7a6cc1e25b97da6b156ad79954dccdc8e3","source":{"kind":"arxiv","id":"2405.16908","version":2},"attestation_state":"computed","paper":{"title":"Can Large Language Models Faithfully Express Their Intrinsic Uncertainty in Words?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Gal Yona, Mor Geva, Roee Aharoni","submitted_at":"2024-05-27T07:56:23Z","abstract_excerpt":"We posit that large language models (LLMs) should be capable of expressing their intrinsic uncertainty in natural language. For example, if the LLM is equally likely to output two contradicting answers to the same question, then its generated response should reflect this uncertainty by hedging its answer (e.g., \"I'm not sure, but I think...\"). We formalize faithful response uncertainty based on the gap between the model's intrinsic confidence in the assertions it makes and the decisiveness by which they are conveyed. This example-level metric reliably indicates whether the model reflects its u"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.16908","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-05-27T07:56:23Z","cross_cats_sorted":[],"title_canon_sha256":"b208d6fcace7fd11a9dba4a8173718b4c55d1f3fbd6ffa75c990023ae63fcbfd","abstract_canon_sha256":"2bf8468c236b34846f0ce486c28985000ca7e41cbd84843886befa67f007d86d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:12:06.579345Z","signature_b64":"Y7RpACzo/9R/2ueyJNuO7P2RYvDfAGEhzqq35QmQGEzQu5TW0eo/QPzAI0Ya3t/Cz0qZt6toun6JAlIC+rB+AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a20117609747aad7795f5ef813690a7a6cc1e25b97da6b156ad79954dccdc8e3","last_reissued_at":"2026-07-05T09:12:06.578829Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:12:06.578829Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Can Large Language Models Faithfully Express Their Intrinsic Uncertainty in Words?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Gal Yona, Mor Geva, Roee Aharoni","submitted_at":"2024-05-27T07:56:23Z","abstract_excerpt":"We posit that large language models (LLMs) should be capable of expressing their intrinsic uncertainty in natural language. For example, if the LLM is equally likely to output two contradicting answers to the same question, then its generated response should reflect this uncertainty by hedging its answer (e.g., \"I'm not sure, but I think...\"). We formalize faithful response uncertainty based on the gap between the model's intrinsic confidence in the assertions it makes and the decisiveness by which they are conveyed. This example-level metric reliably indicates whether the model reflects its u"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.16908","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.16908/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.16908","created_at":"2026-07-05T09:12:06.578890+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.16908v2","created_at":"2026-07-05T09:12:06.578890+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.16908","created_at":"2026-07-05T09:12:06.578890+00:00"},{"alias_kind":"pith_short_12","alias_value":"UIAROYEXI6VN","created_at":"2026-07-05T09:12:06.578890+00:00"},{"alias_kind":"pith_short_16","alias_value":"UIAROYEXI6VNO6K7","created_at":"2026-07-05T09:12:06.578890+00:00"},{"alias_kind":"pith_short_8","alias_value":"UIAROYEX","created_at":"2026-07-05T09:12:06.578890+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06327","citing_title":"Estimating Uncertainty from Reasoning: A Large-Scale Study of Multi- and Crosslingual MCQA Performance in LLMs","ref_index":18,"is_internal_anchor":true},{"citing_arxiv_id":"2605.28571","citing_title":"Not All Uncertainty Is Equal: How Uncertainty Granularity Shapes Human Verification in LLM-Assisted Decision Making","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2502.19463","citing_title":"Hedging and Non-Affirmation: Quantifying LLM Alignment on Questions of Human Rights","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12824","citing_title":"Mechanism Plausibility in Generative Agent-Based Modeling","ref_index":87,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12824","citing_title":"Mechanism Plausibility in Generative Agent-Based Modeling","ref_index":87,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00957","citing_title":"\"I Don't Know\" -- Towards Appropriate Trust with Certainty-Aware Retrieval Augmented Generation","ref_index":49,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UIAROYEXI6VNO6K7L34BG2IKPJ","json":"https://pith.science/pith/UIAROYEXI6VNO6K7L34BG2IKPJ.json","graph_json":"https://pith.science/api/pith-number/UIAROYEXI6VNO6K7L34BG2IKPJ/graph.json","events_json":"https://pith.science/api/pith-number/UIAROYEXI6VNO6K7L34BG2IKPJ/events.json","paper":"https://pith.science/paper/UIAROYEX"},"agent_actions":{"view_html":"https://pith.science/pith/UIAROYEXI6VNO6K7L34BG2IKPJ","download_json":"https://pith.science/pith/UIAROYEXI6VNO6K7L34BG2IKPJ.json","view_paper":"https://pith.science/paper/UIAROYEX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.16908&json=true","fetch_graph":"https://pith.science/api/pith-number/UIAROYEXI6VNO6K7L34BG2IKPJ/graph.json","fetch_events":"https://pith.science/api/pith-number/UIAROYEXI6VNO6K7L34BG2IKPJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UIAROYEXI6VNO6K7L34BG2IKPJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UIAROYEXI6VNO6K7L34BG2IKPJ/action/storage_attestation","attest_author":"https://pith.science/pith/UIAROYEXI6VNO6K7L34BG2IKPJ/action/author_attestation","sign_citation":"https://pith.science/pith/UIAROYEXI6VNO6K7L34BG2IKPJ/action/citation_signature","submit_replication":"https://pith.science/pith/UIAROYEXI6VNO6K7L34BG2IKPJ/action/replication_record"}},"created_at":"2026-07-05T09:12:06.578890+00:00","updated_at":"2026-07-05T09:12:06.578890+00:00"}