{"as_of":"2026-08-20T16:12:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:6aecc38f3513d5d4066ead0f67eecc8dc536e991ab61310885b628b7dd3684f8","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":14,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":14,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-20T06:33:59.587034+00:00","state":"measured"},{"denominator":14,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":14,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T20:38:34.500200Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T20:18:56.663568Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2406.13261","last_updated":"2024-07-08T18:29:58Z","snapshot_observed_at":"2026-08-16T13:41:34.849705Z","submitted_at":"2024-06-19T06:46:59Z","title":"BeHonest: Benchmarking Honesty in Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13261","snapshot_observed_at":"2026-08-07T14:09:50.708278Z","title":"arXiv preprint arXiv:2406.13261","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19806","last_updated":"2025-05-26T10:40:52Z","snapshot_observed_at":"2026-08-18T08:25:32.240298Z","submitted_at":"2025-05-26T10:40:52Z","title":"Exploring Consciousness in LLMs: A Systematic Survey of Theories, Implementations, and Frontier Risks","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-07T14:09:50.708278Z"},"links":{"cited_paper":"/paper/2406.13261","citing_paper":"/paper/2505.19806"},"observation_digest":"sha256:ea7357a35b9cf22fee45e7b7bfb208afffa25b7f7ac9fa578d7479b8426eb2b7","observation_id":"1cc81a92-98cc-4197-9b84-da0c2dcf4763","resolution":{"observed_at":"2026-08-07T14:09:50.708278Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13261","last_updated":"2024-07-08T18:29:58Z","snapshot_observed_at":"2026-08-16T13:41:34.849705Z","submitted_at":"2024-06-19T06:46:59Z","title":"BeHonest: Benchmarking Honesty in Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13261","snapshot_observed_at":"2026-08-07T11:50:53.798081Z","title":"Behonest: Benchmarking honesty in large language models.arXiv preprint arXiv:2406.13261, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01252","last_updated":"2025-06-02T02:01:40Z","snapshot_observed_at":"2026-08-18T14:35:47.158614Z","submitted_at":"2025-06-02T02:01:40Z","title":"MTCMB: A Multi-Task Benchmark Framework for Evaluating LLMs on Knowledge, Reasoning, and Safety in Traditional Chinese Medicine","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T11:50:53.798081Z"},"links":{"cited_paper":"/paper/2406.13261","citing_paper":"/paper/2506.01252"},"observation_digest":"sha256:77456ab93a0874519ab8fbde73d76e513abb726d33bec99d30a1672ffc88817e","observation_id":"4f920a70-598d-475b-82ab-86009dc57379","resolution":{"observed_at":"2026-08-07T11:50:53.798081Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13261","last_updated":"2024-07-08T18:29:58Z","snapshot_observed_at":"2026-08-16T13:41:34.849705Z","submitted_at":"2024-06-19T06:46:59Z","title":"BeHonest: Benchmarking Honesty in Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13261","snapshot_observed_at":"2026-08-06T19:43:58.189681Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.04748","last_updated":"2025-07-07T08:19:17Z","snapshot_observed_at":"2026-08-17T02:48:24.509682Z","submitted_at":"2025-07-07T08:19:17Z","title":"LLM-based Question-Answer Framework for Sensor-driven HVAC System Interaction","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:58.189681Z"},"links":{"cited_paper":"/paper/2406.13261","citing_paper":"/paper/2507.04748"},"observation_digest":"sha256:3b4400bec162d61a8d351c5159f771cb3c5fa1fea6041bdf1c1177045dea23a4","observation_id":"716eb961-9bcf-48e3-8049-2dbe0bfc9589","resolution":{"observed_at":"2026-08-06T19:43:58.189681Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13261","last_updated":"2024-07-08T18:29:58Z","snapshot_observed_at":"2026-08-16T13:41:34.849705Z","submitted_at":"2024-06-19T06:46:59Z","title":"BeHonest: Benchmarking Honesty in Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13261","snapshot_observed_at":"2026-08-06T14:13:06.291665Z","title":"Behonest: Benchmarking honesty in large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19672","last_updated":"2025-07-25T20:52:58Z","snapshot_observed_at":"2026-08-07T12:02:14.124506Z","submitted_at":"2025-07-25T20:52:58Z","title":"Alignment and Safety in Large Language Models: Safety Mechanisms, Training Paradigms, and Emerging Challenges","version":1},"reference_index":98,"source":"arxiv_source","source_observed_at":"2026-08-06T14:13:06.291665Z"},"links":{"cited_paper":"/paper/2406.13261","citing_paper":"/paper/2507.19672"},"observation_digest":"sha256:501e7751f99563aa0c7981297531d67477702be3a34decdee8fdec16deff824c","observation_id":"43f927cd-3bec-473e-9032-bf525bfac3e5","resolution":{"observed_at":"2026-08-06T14:13:06.291665Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13261","last_updated":"2024-07-08T18:29:58Z","snapshot_observed_at":"2026-08-16T13:41:34.849705Z","submitted_at":"2024-06-19T06:46:59Z","title":"BeHonest: Benchmarking Honesty in Large Language Models","version":3},"cited_work":{"arxiv_id":"2406.13261","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13261","snapshot_observed_at":"2026-07-03T20:18:56.663568Z","title":"Behonest: Benchmarking honesty in large language models","venue":null,"work_id":"88ad184d-4abc-44ad-930f-2785d37bfc8f","year":2025},"citing_paper":{"arxiv_id":"2605.08765","last_updated":"2026-05-09T07:50:27Z","snapshot_observed_at":"2026-08-12T21:11:22.212777Z","submitted_at":"2026-05-09T07:50:27Z","title":"Unlearners Can Lie: Evaluating and Improving Honesty in LLM Unlearning","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-12T02:47:10.818363Z"},"links":{"cited_paper":"/paper/2406.13261","citing_paper":"/paper/2605.08765"},"observation_digest":"sha256:f5fceabd6b8fd4a8c650cba1e86d4be602ee1afe9cb16e3e3639e57f7dade51b","observation_id":"75180596-08b1-4632-8046-36a663f80dea","resolution":{"observed_at":"2026-05-12T07:31:24.489114Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13261","last_updated":"2024-07-08T18:29:58Z","snapshot_observed_at":"2026-08-16T13:41:34.849705Z","submitted_at":"2024-06-19T06:46:59Z","title":"BeHonest: Benchmarking Honesty in Large Language Models","version":3},"cited_work":{"arxiv_id":"2406.13261","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13261","snapshot_observed_at":"2026-07-03T20:18:56.663568Z","title":"Behonest: Benchmarking honesty in large language models","venue":null,"work_id":"88ad184d-4abc-44ad-930f-2785d37bfc8f","year":2025},"citing_paper":{"arxiv_id":"2605.10639","last_updated":"2026-05-11T14:27:39Z","snapshot_observed_at":"2026-08-15T07:47:05.762692Z","submitted_at":"2026-05-11T14:27:39Z","title":"Navigating the Sea of LLM Evaluation: Investigating Bias in Toxicity Benchmarks","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-12T05:28:45.453455Z"},"links":{"cited_paper":"/paper/2406.13261","citing_paper":"/paper/2605.10639"},"observation_digest":"sha256:62c2cf2899018eb7302928b441a0b90ffb8e478625cf38b15835c4514b3c1198","observation_id":"13d213e0-542c-436b-8658-ed6f6f6dc5d4","resolution":{"observed_at":"2026-05-12T05:31:23.935337Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13261","last_updated":"2024-07-08T18:29:58Z","snapshot_observed_at":"2026-08-16T13:41:34.849705Z","submitted_at":"2024-06-19T06:46:59Z","title":"BeHonest: Benchmarking Honesty in Large Language Models","version":3},"cited_work":{"arxiv_id":"2406.13261","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13261","snapshot_observed_at":"2026-07-03T20:18:56.663568Z","title":"Behonest: Benchmarking honesty in large language models","venue":null,"work_id":"88ad184d-4abc-44ad-930f-2785d37bfc8f","year":2025},"citing_paper":{"arxiv_id":"2605.19270","last_updated":"2026-05-19T02:33:21Z","snapshot_observed_at":"2026-08-16T08:19:44.478984Z","submitted_at":"2026-05-19T02:33:21Z","title":"DECOR: Auditing LLM Deception via Information Manipulation Theory","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-20T06:27:10.445757Z"},"links":{"cited_paper":"/paper/2406.13261","citing_paper":"/paper/2605.19270"},"observation_digest":"sha256:17066c7cd74e885da6f8e2810ef365f24afbb783e4e0269429e4b2de5670b361","observation_id":"f48f2433-9f2c-487e-91e2-55c73e13ae0e","resolution":{"observed_at":"2026-05-20T06:28:05.381264Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13261","last_updated":"2024-07-08T18:29:58Z","snapshot_observed_at":"2026-08-16T13:41:34.849705Z","submitted_at":"2024-06-19T06:46:59Z","title":"BeHonest: Benchmarking Honesty in Large Language Models","version":3},"cited_work":{"arxiv_id":"2406.13261","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13261","snapshot_observed_at":"2026-07-03T20:18:56.663568Z","title":"Behonest: Benchmarking honesty in large language models","venue":null,"work_id":"88ad184d-4abc-44ad-930f-2785d37bfc8f","year":2025},"citing_paper":{"arxiv_id":"2606.02380","last_updated":"2026-06-28T14:28:39Z","snapshot_observed_at":"2026-08-13T13:34:13.036756Z","submitted_at":"2026-06-01T15:28:34Z","title":"SPADE-Bench: Evaluating Spontaneous Strategic Deception in Agents via Plan-Action Divergence","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-28T14:39:51.672024Z"},"links":{"cited_paper":"/paper/2406.13261","citing_paper":"/paper/2606.02380"},"observation_digest":"sha256:a66089f6232a5bf1d1b29d8e3e547cefda7936a896f4a74cfcdf10ea435356c4","observation_id":"bf0bdbc0-8745-477a-894d-c71c971ee0a0","resolution":{"observed_at":"2026-07-01T23:06:20.590107Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13261","last_updated":"2024-07-08T18:29:58Z","snapshot_observed_at":"2026-08-16T13:41:34.849705Z","submitted_at":"2024-06-19T06:46:59Z","title":"BeHonest: Benchmarking Honesty in Large Language Models","version":3},"cited_work":{"arxiv_id":"2406.13261","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13261","snapshot_observed_at":"2026-07-03T20:18:56.663568Z","title":"Behonest: Benchmarking honesty in large language models","venue":null,"work_id":"88ad184d-4abc-44ad-930f-2785d37bfc8f","year":2025},"citing_paper":{"arxiv_id":"2606.02380","last_updated":"2026-06-28T14:28:39Z","snapshot_observed_at":"2026-08-13T13:34:13.036756Z","submitted_at":"2026-06-01T15:28:34Z","title":"SPADE-Bench: Evaluating Spontaneous Strategic Deception in Agents via Plan-Action Divergence","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-30T10:45:44.127391Z"},"links":{"cited_paper":"/paper/2406.13261","citing_paper":"/paper/2606.02380"},"observation_digest":"sha256:0416a9d8f0caf51c2325f42e1728890f4c2044d8bb951bcc48c4623e686d8bc3","observation_id":"cfbb8b37-c1c3-4263-bdd6-9f2c03bfb24b","resolution":{"observed_at":"2026-06-30T10:54:36.985886Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13261","last_updated":"2024-07-08T18:29:58Z","snapshot_observed_at":"2026-08-16T13:41:34.849705Z","submitted_at":"2024-06-19T06:46:59Z","title":"BeHonest: Benchmarking Honesty in Large Language Models","version":3},"cited_work":{"arxiv_id":"2406.13261","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13261","snapshot_observed_at":"2026-07-03T20:18:56.663568Z","title":"Behonest: Benchmarking honesty in large language models","venue":null,"work_id":"88ad184d-4abc-44ad-930f-2785d37bfc8f","year":2025},"citing_paper":{"arxiv_id":"2606.13310","last_updated":"2026-06-11T13:07:02Z","snapshot_observed_at":"2026-08-15T02:53:25.014777Z","submitted_at":"2026-06-11T13:07:02Z","title":"RogueAI: A Reverse Turing Test for Detecting Licensed AI Deception in Dialogue","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-06-27T06:34:39.457798Z"},"links":{"cited_paper":"/paper/2406.13261","citing_paper":"/paper/2606.13310"},"observation_digest":"sha256:aaf10ebaa9ab434d646822f6da9a2fde3d49fc15f4ea28b76d912e8358533990","observation_id":"d33f246a-485b-4c20-9b3e-204721c35d77","resolution":{"observed_at":"2026-07-03T15:18:33.688342Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13261","last_updated":"2024-07-08T18:29:58Z","snapshot_observed_at":"2026-08-16T13:41:34.849705Z","submitted_at":"2024-06-19T06:46:59Z","title":"BeHonest: Benchmarking Honesty in Large Language Models","version":3},"cited_work":{"arxiv_id":"2406.13261","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13261","snapshot_observed_at":"2026-07-03T20:18:56.663568Z","title":"Behonest: Benchmarking honesty in large language models","venue":null,"work_id":"88ad184d-4abc-44ad-930f-2785d37bfc8f","year":2025},"citing_paper":{"arxiv_id":"2606.17478","last_updated":"2026-06-16T03:41:29Z","snapshot_observed_at":"2026-08-14T19:39:32.564410Z","submitted_at":"2026-06-16T03:41:29Z","title":"Decoding Hidden Deception in Reasoning LLMs: Activation Explainers for Deception Auditing","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-27T01:28:38.810889Z"},"links":{"cited_paper":"/paper/2406.13261","citing_paper":"/paper/2606.17478"},"observation_digest":"sha256:e07b2832a71f0723f8ff9512bd7d6d73dbe92687006414c94079fb5186b106fc","observation_id":"05a3b2ca-af7e-45c6-bd80-db214128ad5e","resolution":{"observed_at":"2026-07-03T20:18:56.665646Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13261","last_updated":"2024-07-08T18:29:58Z","snapshot_observed_at":"2026-08-16T13:41:34.849705Z","submitted_at":"2024-06-19T06:46:59Z","title":"BeHonest: Benchmarking Honesty in Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13261","snapshot_observed_at":"2026-08-11T04:15:50.166929Z","title":"arXiv preprint arXiv:2406.13261 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-20T07:04:46.927279Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.166929Z"},"links":{"cited_paper":"/paper/2406.13261","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:796078bbf538411cb79eb06119fbd1d32fd92835a5068a3b1fdc1807bbe63f96","observation_id":"fc86b4ff-f50e-443f-b78b-adfa071959e4","resolution":{"observed_at":"2026-08-11T04:15:50.166929Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13261","last_updated":"2024-07-08T18:29:58Z","snapshot_observed_at":"2026-08-16T13:41:34.849705Z","submitted_at":"2024-06-19T06:46:59Z","title":"BeHonest: Benchmarking Honesty in Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13261","snapshot_observed_at":"2026-08-15T20:38:34.500200Z","title":"arXiv preprint arXiv:2406.13261 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.12921","last_updated":"2026-08-14T04:21:23Z","snapshot_observed_at":"2026-08-19T23:09:30.732686Z","submitted_at":"2026-08-13T08:03:02Z","title":"Discovering Efficient and Explainable Communication Topologies for LLM-based Multi-Agent Systems via Causal Inference","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-15T20:38:34.500200Z"},"links":{"cited_paper":"/paper/2406.13261","citing_paper":"/paper/2608.12921"},"observation_digest":"sha256:26ed86f01ab082e7a61caeb29b8ffed9b570a4b9fdc65a739e70fb345f33a504","observation_id":"fc96404e-bfe2-44c4-a015-a4e9f1703cff","resolution":{"observed_at":"2026-08-15T20:38:34.500200Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13261","last_updated":"2024-07-08T18:29:58Z","snapshot_observed_at":"2026-08-16T13:41:34.849705Z","submitted_at":"2024-06-19T06:46:59Z","title":"BeHonest: Benchmarking Honesty in Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13261","snapshot_observed_at":"2026-08-14T15:06:26.693276Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.13267","last_updated":"2026-08-13T14:06:35Z","snapshot_observed_at":"2026-08-18T09:22:33.180564Z","submitted_at":"2026-08-13T14:06:35Z","title":"How Do VLMs Behave When Blind or Misled? Behavioral Evaluation of VLMs on Scientific Figures","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-14T15:06:26.693276Z"},"links":{"cited_paper":"/paper/2406.13261","citing_paper":"/paper/2608.13267"},"observation_digest":"sha256:90f356d28936682eb18df766bc9f85821233915dd85a64741bb2ca152a82c9ef","observation_id":"4ad81314-3d72-4c7c-b999-843c4ecb3b3c","resolution":{"observed_at":"2026-08-14T15:06:26.693276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2406.13261/citation-record","integrity":"/paper/2406.13261/integrity","json":"/paper/2406.13261/citation-record.json","paper":"/paper/2406.13261"},"outbound":[],"paper":{"arxiv_id":"2406.13261","last_updated":"2024-07-08T18:29:58Z","latest_version":3,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-16T13:41:34.849705Z","submitted_at":"2024-06-19T06:46:59Z","title":"BeHonest: Benchmarking Honesty in Large Language Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"thesis":"As of 20 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 14 inbound Pith citation observations for arXiv:2406.13261."}