{"as_of":"2026-08-08T13:52:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e2154dc16b15ae7020424ad910402b71ceecc3d6049c3a0523d8e03be2196ecb","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":11,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":11,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":11,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":11,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:09:36.447956Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-02T21:07:23.912546Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2402.13213","last_updated":"2025-08-07T01:33:12Z","snapshot_observed_at":"2026-08-03T14:34:50.040904Z","submitted_at":"2024-02-20T18:24:47Z","title":"Probabilities of Chat LLMs Are Miscalibrated but Still Predict Correctness on Multiple-Choice Q&A","version":4},"cited_work":{"arxiv_id":"2402.13213","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.13213","snapshot_observed_at":"2026-07-02T21:07:23.912546Z","title":"Softmax probabilities (mostly) predict large language model correctness on multiple-choice q&a","venue":null,"work_id":"d46c9875-4f26-41fd-8e58-2bbb5f1f8276","year":2024},"citing_paper":{"arxiv_id":"2412.02904","last_updated":"2026-04-19T22:48:18Z","snapshot_observed_at":"2026-08-02T22:43:50.593390Z","submitted_at":"2024-12-03T23:14:47Z","title":"Enhancing Trust in Large Language Models via Uncertainty-Calibrated Fine-Tuning","version":2},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-05-23T07:45:50.292586Z"},"links":{"cited_paper":"/paper/2402.13213","citing_paper":"/paper/2412.02904"},"observation_digest":"sha256:31d6632eac77faf02351369222671fdbc91e92e9ebcaeb7568dffc9ae3a13339","observation_id":"6b3a3b5c-f38f-44e4-8fee-2de7ebce61bd","resolution":{"observed_at":"2026-05-23T07:47:42.669613Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.13213","last_updated":"2025-08-07T01:33:12Z","snapshot_observed_at":"2026-08-03T14:34:50.040904Z","submitted_at":"2024-02-20T18:24:47Z","title":"Probabilities of Chat LLMs Are Miscalibrated but Still Predict Correctness on Multiple-Choice Q&A","version":4},"cited_work":{"arxiv_id":"2402.13213","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.13213","snapshot_observed_at":"2026-07-02T21:07:23.912546Z","title":"Softmax probabilities (mostly) predict large language model correctness on multiple-choice q&a","venue":null,"work_id":"d46c9875-4f26-41fd-8e58-2bbb5f1f8276","year":2024},"citing_paper":{"arxiv_id":"2412.15176","last_updated":"2026-04-20T11:16:26Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T18:51:06Z","title":"Rethinking Uncertainty Estimation in LLMs: A Principled Single-Sequence Measure","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-23T06:22:25.496269Z"},"links":{"cited_paper":"/paper/2402.13213","citing_paper":"/paper/2412.15176"},"observation_digest":"sha256:b68b96450e0c0f91de23af858ee1ea679b1d6f2b1f7e2d8c7d05fd087c848e35","observation_id":"78a6a99f-2285-48aa-817a-140b6cece409","resolution":{"observed_at":"2026-05-23T06:22:38.186542Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.13213","last_updated":"2025-08-07T01:33:12Z","snapshot_observed_at":"2026-08-03T14:34:50.040904Z","submitted_at":"2024-02-20T18:24:47Z","title":"Probabilities of Chat LLMs Are Miscalibrated but Still Predict Correctness on Multiple-Choice Q&A","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.13213","snapshot_observed_at":"2026-08-07T15:09:36.447956Z","title":"Softmax Probabiliti es (Mostly) Predict Large Language Model Correctness on Multiple-Choi ce Q&A,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21518","last_updated":"2025-05-22T05:59:36Z","snapshot_observed_at":"2026-08-07T15:02:21.299946Z","submitted_at":"2025-05-22T05:59:36Z","title":"Resilient LLM-Empowered Semantic MAC Protocols via Zero-Shot Adaptation and Knowledge Distillation","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T15:09:36.447956Z"},"links":{"cited_paper":"/paper/2402.13213","citing_paper":"/paper/2505.21518"},"observation_digest":"sha256:d4f605aae617cd2a3524695572743e5da8a208357daaac0b8ac9cc91ecc23b51","observation_id":"d77c015c-0fb6-4b89-81cc-ca57c5ff6e60","resolution":{"observed_at":"2026-08-07T15:09:36.447956Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.13213","last_updated":"2025-08-07T01:33:12Z","snapshot_observed_at":"2026-08-03T14:34:50.040904Z","submitted_at":"2024-02-20T18:24:47Z","title":"Probabilities of Chat LLMs Are Miscalibrated but Still Predict Correctness on Multiple-Choice Q&A","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.13213","snapshot_observed_at":"2026-08-06T12:55:57.711866Z","title":"Association for Computational Linguistics","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.21406","last_updated":"2025-07-29T00:26:33Z","snapshot_observed_at":"2026-08-06T12:55:56.526336Z","submitted_at":"2025-07-29T00:26:33Z","title":"Shapley Uncertainty in Natural Language Generation","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-06T12:55:57.711866Z"},"links":{"cited_paper":"/paper/2402.13213","citing_paper":"/paper/2507.21406"},"observation_digest":"sha256:619a81a9151f74652d1dff467ba32ece0bb35c7a580e8cc9c54adc72d5ae7f67","observation_id":"48e9aed8-a786-4985-8e97-ca104f4f2d9b","resolution":{"observed_at":"2026-08-06T12:55:57.711866Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.13213","last_updated":"2025-08-07T01:33:12Z","snapshot_observed_at":"2026-08-03T14:34:50.040904Z","submitted_at":"2024-02-20T18:24:47Z","title":"Probabilities of Chat LLMs Are Miscalibrated but Still Predict Correctness on Multiple-Choice Q&A","version":4},"cited_work":{"arxiv_id":"2402.13213","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.13213","snapshot_observed_at":"2026-07-02T21:07:23.912546Z","title":"Softmax probabilities (mostly) predict large language model correctness on multiple-choice q&a","venue":null,"work_id":"d46c9875-4f26-41fd-8e58-2bbb5f1f8276","year":2024},"citing_paper":{"arxiv_id":"2603.22161","last_updated":"2026-05-19T14:46:51Z","snapshot_observed_at":"2026-08-06T02:25:28.537505Z","submitted_at":"2026-03-23T16:23:31Z","title":"Causal Evidence that Language Models use Confidence to Drive Behavior","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-21T09:43:05.524088Z"},"links":{"cited_paper":"/paper/2402.13213","citing_paper":"/paper/2603.22161"},"observation_digest":"sha256:c113601156c70edb3beeffa5fb40808387c0ad333b6a34deca87bbc84f8d624d","observation_id":"62e24b5d-6bee-4a4e-b606-6396c400275f","resolution":{"observed_at":"2026-05-21T09:44:05.679600Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.13213","last_updated":"2025-08-07T01:33:12Z","snapshot_observed_at":"2026-08-03T14:34:50.040904Z","submitted_at":"2024-02-20T18:24:47Z","title":"Probabilities of Chat LLMs Are Miscalibrated but Still Predict Correctness on Multiple-Choice Q&A","version":4},"cited_work":{"arxiv_id":"2402.13213","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.13213","snapshot_observed_at":"2026-07-02T21:07:23.912546Z","title":"Softmax probabilities (mostly) predict large language model correctness on multiple-choice q&a","venue":null,"work_id":"d46c9875-4f26-41fd-8e58-2bbb5f1f8276","year":2024},"citing_paper":{"arxiv_id":"2604.17200","last_updated":"2026-04-19T02:04:14Z","snapshot_observed_at":"2026-07-06T23:04:23.730478Z","submitted_at":"2026-04-19T02:04:14Z","title":"Calibrating Model-Based Evaluation Metrics for Summarization","version":1},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-05-10T06:36:55.334742Z"},"links":{"cited_paper":"/paper/2402.13213","citing_paper":"/paper/2604.17200"},"observation_digest":"sha256:efd26b1df9c6ed6f04e0a2ed6e413e581d5392b2ebd41260f55911c1b19f3386","observation_id":"41208fd2-6bb0-40be-981e-f0c1201f221c","resolution":{"observed_at":"2026-05-10T06:41:37.021662Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.13213","last_updated":"2025-08-07T01:33:12Z","snapshot_observed_at":"2026-08-03T14:34:50.040904Z","submitted_at":"2024-02-20T18:24:47Z","title":"Probabilities of Chat LLMs Are Miscalibrated but Still Predict Correctness on Multiple-Choice Q&A","version":4},"cited_work":{"arxiv_id":"2402.13213","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.13213","snapshot_observed_at":"2026-07-02T21:07:23.912546Z","title":"Softmax probabilities (mostly) predict large language model correctness on multiple-choice q&a","venue":null,"work_id":"d46c9875-4f26-41fd-8e58-2bbb5f1f8276","year":2024},"citing_paper":{"arxiv_id":"2605.09273","last_updated":"2026-05-21T01:08:31Z","snapshot_observed_at":"2026-08-07T18:58:24.921148Z","submitted_at":"2026-05-10T02:45:59Z","title":"Instance-Adaptive Online Multicalibration","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-05-12T04:45:02.146222Z"},"links":{"cited_paper":"/paper/2402.13213","citing_paper":"/paper/2605.09273"},"observation_digest":"sha256:ff45e551576f2f0460b364409e9a430cf8c6fe0a2d98c5b3951abce1941d23d0","observation_id":"a2506d7f-6ae5-4966-9f74-774be58d8d5c","resolution":{"observed_at":"2026-05-12T05:56:40.742330Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.13213","last_updated":"2025-08-07T01:33:12Z","snapshot_observed_at":"2026-08-03T14:34:50.040904Z","submitted_at":"2024-02-20T18:24:47Z","title":"Probabilities of Chat LLMs Are Miscalibrated but Still Predict Correctness on Multiple-Choice Q&A","version":4},"cited_work":{"arxiv_id":"2402.13213","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.13213","snapshot_observed_at":"2026-07-02T21:07:23.912546Z","title":"Softmax probabilities (mostly) predict large language model correctness on multiple-choice q&a","venue":null,"work_id":"d46c9875-4f26-41fd-8e58-2bbb5f1f8276","year":2024},"citing_paper":{"arxiv_id":"2605.09273","last_updated":"2026-05-21T01:08:31Z","snapshot_observed_at":"2026-08-07T18:58:24.921148Z","submitted_at":"2026-05-10T02:45:59Z","title":"Instance-Adaptive Online Multicalibration","version":2},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-05-22T09:41:09.689528Z"},"links":{"cited_paper":"/paper/2402.13213","citing_paper":"/paper/2605.09273"},"observation_digest":"sha256:e09e66e5852b30e8aada7f5396a05a8a9578b454d462a37ec83cb4e258bf4b95","observation_id":"8682f582-6878-4837-bdb2-f5c8047ba853","resolution":{"observed_at":"2026-05-22T09:41:21.584990Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.13213","last_updated":"2025-08-07T01:33:12Z","snapshot_observed_at":"2026-08-03T14:34:50.040904Z","submitted_at":"2024-02-20T18:24:47Z","title":"Probabilities of Chat LLMs Are Miscalibrated but Still Predict Correctness on Multiple-Choice Q&A","version":4},"cited_work":{"arxiv_id":"2402.13213","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.13213","snapshot_observed_at":"2026-07-02T21:07:23.912546Z","title":"Softmax probabilities (mostly) predict large language model correctness on multiple-choice q&a","venue":null,"work_id":"d46c9875-4f26-41fd-8e58-2bbb5f1f8276","year":2024},"citing_paper":{"arxiv_id":"2605.10202","last_updated":"2026-05-11T08:48:04Z","snapshot_observed_at":"2026-07-06T23:22:13.774652Z","submitted_at":"2026-05-11T08:48:04Z","title":"Task-Aware Calibration: Provably Optimal Decoding in LLMs","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-12T02:55:54.933129Z"},"links":{"cited_paper":"/paper/2402.13213","citing_paper":"/paper/2605.10202"},"observation_digest":"sha256:1e692dff222de79c8fb4022edc5dc2b43c30ec9c1c8c25b7fe2363515535e7eb","observation_id":"ca86e480-9f27-48d4-84b8-949bc233cd15","resolution":{"observed_at":"2026-05-12T02:56:18.701791Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.13213","last_updated":"2025-08-07T01:33:12Z","snapshot_observed_at":"2026-08-03T14:34:50.040904Z","submitted_at":"2024-02-20T18:24:47Z","title":"Probabilities of Chat LLMs Are Miscalibrated but Still Predict Correctness on Multiple-Choice Q&A","version":4},"cited_work":{"arxiv_id":"2402.13213","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.13213","snapshot_observed_at":"2026-07-02T21:07:23.912546Z","title":"Softmax probabilities (mostly) predict large language model correctness on multiple-choice q&a","venue":null,"work_id":"d46c9875-4f26-41fd-8e58-2bbb5f1f8276","year":2024},"citing_paper":{"arxiv_id":"2605.21776","last_updated":"2026-05-20T22:10:54Z","snapshot_observed_at":"2026-08-06T21:55:06.899213Z","submitted_at":"2026-05-20T22:10:54Z","title":"PromptNCE: Pointwise Mutual Information Predictions Using Only LLMs and Contrastive Estimation Prompts","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-05-22T08:46:12.102599Z"},"links":{"cited_paper":"/paper/2402.13213","citing_paper":"/paper/2605.21776"},"observation_digest":"sha256:21573243c8cf181210536f95f7d539e50efa200943a622f77c9388fd195ca8f7","observation_id":"fc9ccb72-0e16-4a51-a59c-c0eda646e3c9","resolution":{"observed_at":"2026-05-22T08:46:18.168776Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.13213","last_updated":"2025-08-07T01:33:12Z","snapshot_observed_at":"2026-08-03T14:34:50.040904Z","submitted_at":"2024-02-20T18:24:47Z","title":"Probabilities of Chat LLMs Are Miscalibrated but Still Predict Correctness on Multiple-Choice Q&A","version":4},"cited_work":{"arxiv_id":"2402.13213","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.13213","snapshot_observed_at":"2026-07-02T21:07:23.912546Z","title":"Softmax probabilities (mostly) predict large language model correctness on multiple-choice q&a","venue":null,"work_id":"d46c9875-4f26-41fd-8e58-2bbb5f1f8276","year":2024},"citing_paper":{"arxiv_id":"2606.08088","last_updated":"2026-06-06T10:23:24Z","snapshot_observed_at":"2026-07-06T23:47:38.671681Z","submitted_at":"2026-06-06T10:23:24Z","title":"ConSteer-RL: Steering Reasoning Capabilities in Large Language Models via Confidence-Aware Reinforcement Learning","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-06-27T19:56:09.820812Z"},"links":{"cited_paper":"/paper/2402.13213","citing_paper":"/paper/2606.08088"},"observation_digest":"sha256:207d5acab4302487f0f579fa2dfedcc2b086d754cd58624f40b3e36305e11fdd","observation_id":"96734832-56cb-42d8-90c9-2cd72e1a68b6","resolution":{"observed_at":"2026-07-02T21:07:23.914209Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2402.13213/citation-record","integrity":"/paper/2402.13213/integrity","json":"/paper/2402.13213/citation-record.json","paper":"/paper/2402.13213"},"outbound":[],"paper":{"arxiv_id":"2402.13213","last_updated":"2025-08-07T01:33:12Z","latest_version":4,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-03T14:34:50.040904Z","submitted_at":"2024-02-20T18:24:47Z","title":"Probabilities of Chat LLMs Are Miscalibrated but Still Predict Correctness on Multiple-Choice Q&A"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 11 inbound Pith citation observations for arXiv:2402.13213."}