{"as_of":"2026-08-14T12:56:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:d0fb5d956ae9ea3fa800f257135cbfc1ef3107ea4b695aa6f96494978c48ba7b","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":6,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":6,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-14T06:32:32.682623+00:00","state":"measured"},{"denominator":6,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":6,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T12:38:06.553485Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-06-30T09:54:34.363302Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2311.15930","last_updated":"2023-11-27T15:38:17Z","snapshot_observed_at":"2026-08-13T05:16:56.390928Z","submitted_at":"2023-11-27T15:38:17Z","title":"WorldSense: A Synthetic Benchmark for Grounded Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.15930","snapshot_observed_at":"2026-08-11T12:38:06.553485Z","title":"Worldsense: A synthetic benchmark for grounded reasoning in large language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13989","last_updated":"2024-12-18T16:09:42Z","snapshot_observed_at":"2026-08-12T01:33:55.803461Z","submitted_at":"2024-12-18T16:09:42Z","title":"What makes a good metric? Evaluating automatic metrics for text-to-image consistency","version":1},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-11T12:38:06.553485Z"},"links":{"cited_paper":"/paper/2311.15930","citing_paper":"/paper/2412.13989"},"observation_digest":"sha256:85c72d2d865dca6d38bc7497c471291c39dc8c229eecc8de4202d6a876ecda67","observation_id":"f8f0eff5-13ee-4c79-8580-69b6724a0d57","resolution":{"observed_at":"2026-08-11T12:38:06.553485Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.15930","last_updated":"2023-11-27T15:38:17Z","snapshot_observed_at":"2026-08-13T05:16:56.390928Z","submitted_at":"2023-11-27T15:38:17Z","title":"WorldSense: A Synthetic Benchmark for Grounded Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.15930","snapshot_observed_at":"2026-08-07T05:01:08.270084Z","title":"WorldSense: A Synthetic Benchmark for Grounded Reasoning in Large Language Models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.09038","last_updated":"2025-06-10T17:57:30Z","snapshot_observed_at":"2026-08-13T13:49:09.715847Z","submitted_at":"2025-06-10T17:57:30Z","title":"AbstentionBench: Reasoning LLMs Fail on Unanswerable Questions","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-07T05:01:08.270084Z"},"links":{"cited_paper":"/paper/2311.15930","citing_paper":"/paper/2506.09038"},"observation_digest":"sha256:2256916908bce37c3ebe9463129f35030dece7c7cb423dbc9a0c063132665c29","observation_id":"d939dd53-b3f1-4aae-adff-335a7b2bd631","resolution":{"observed_at":"2026-08-07T05:01:08.270084Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.15930","last_updated":"2023-11-27T15:38:17Z","snapshot_observed_at":"2026-08-13T05:16:56.390928Z","submitted_at":"2023-11-27T15:38:17Z","title":"WorldSense: A Synthetic Benchmark for Grounded Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.15930","snapshot_observed_at":"2026-08-07T04:44:00.504029Z","title":"Benchekroun, M","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.09849","last_updated":"2025-06-11T15:21:16Z","snapshot_observed_at":"2026-08-12T15:10:44.435246Z","submitted_at":"2025-06-11T15:21:16Z","title":"IntPhys 2: Benchmarking Intuitive Physics Understanding In Complex Synthetic Environments","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T04:44:00.504029Z"},"links":{"cited_paper":"/paper/2311.15930","citing_paper":"/paper/2506.09849"},"observation_digest":"sha256:8382695aae9eb2cb7b0c17503d3b9b9bedbae67d4b63ae860de977cd083015b8","observation_id":"9ee9361e-f6db-4092-97b8-f337a7bbc683","resolution":{"observed_at":"2026-08-07T04:44:00.504029Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.15930","last_updated":"2023-11-27T15:38:17Z","snapshot_observed_at":"2026-08-13T05:16:56.390928Z","submitted_at":"2023-11-27T15:38:17Z","title":"WorldSense: A Synthetic Benchmark for Grounded Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2311.15930","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.15930","snapshot_observed_at":"2026-06-30T09:54:34.363302Z","title":"Worldsense: A synthetic benchmark for grounded reasoning in large language models.arXiv preprint arXiv:2311.15930, 2023","venue":null,"work_id":"3d2fcff3-347f-4064-9e25-ef3d071ff9be","year":2023},"citing_paper":{"arxiv_id":"2606.00869","last_updated":"2026-05-30T19:53:19Z","snapshot_observed_at":"2026-08-13T00:36:06.236099Z","submitted_at":"2026-05-30T19:53:19Z","title":"Enhancing LLM Metacognition via Cognitive Pairwise Training","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-06-28T19:01:18.153145Z"},"links":{"cited_paper":"/paper/2311.15930","citing_paper":"/paper/2606.00869"},"observation_digest":"sha256:71a80c8414f6d29bd49263661e259aacff958cc69828a75aca6c9092132abb72","observation_id":"87a0a974-f900-43b5-b4e8-7a7afb72d8e2","resolution":{"observed_at":"2026-06-28T19:02:33.886009Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.15930","last_updated":"2023-11-27T15:38:17Z","snapshot_observed_at":"2026-08-13T05:16:56.390928Z","submitted_at":"2023-11-27T15:38:17Z","title":"WorldSense: A Synthetic Benchmark for Grounded Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":"2311.15930","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.15930","snapshot_observed_at":"2026-06-30T09:54:34.363302Z","title":"Worldsense: A synthetic benchmark for grounded reasoning in large language models.arXiv preprint arXiv:2311.15930, 2023","venue":null,"work_id":"3d2fcff3-347f-4064-9e25-ef3d071ff9be","year":2023},"citing_paper":{"arxiv_id":"2606.28733","last_updated":"2026-06-27T04:49:37Z","snapshot_observed_at":"2026-08-06T08:47:36.002804Z","submitted_at":"2026-06-27T04:49:37Z","title":"Agentic Abstention: Do Agents Know When to Stop Instead of Act?","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-06-30T09:54:07.138157Z"},"links":{"cited_paper":"/paper/2311.15930","citing_paper":"/paper/2606.28733"},"observation_digest":"sha256:b61092f7f4086bef324942b4bd5883898ed2b12eadfd69cbaf0f2337c50e432d","observation_id":"67e23275-1227-4c0f-9ea2-54dc0da607df","resolution":{"observed_at":"2026-06-30T09:54:34.365047Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.15930","last_updated":"2023-11-27T15:38:17Z","snapshot_observed_at":"2026-08-13T05:16:56.390928Z","submitted_at":"2023-11-27T15:38:17Z","title":"WorldSense: A Synthetic Benchmark for Grounded Reasoning in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.15930","snapshot_observed_at":"2026-07-13T00:53:20.749426Z","title":"Worldsense: A synthetic benchmark for grounded reasoning in large language models.arXiv preprint arXiv:2311.15930, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.09029","last_updated":"2026-07-10T01:31:02Z","snapshot_observed_at":"2026-08-13T12:38:47.987678Z","submitted_at":"2026-07-10T01:31:02Z","title":"MOSAIC: Adaptive Inter-layer Composition for Efficient Heterogeneous Vision-Language Models","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-07-13T00:53:20.749426Z"},"links":{"cited_paper":"/paper/2311.15930","citing_paper":"/paper/2607.09029"},"observation_digest":"sha256:84c2c92aa45c6e160c3e33be1af8cc0f1a57feae03fda5f5156f4031ee01c8f6","observation_id":"f46d6a71-f30f-4a3e-a6e9-16b47e02561d","resolution":{"observed_at":"2026-07-13T00:53:20.749426Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2311.15930/citation-record","integrity":"/paper/2311.15930/integrity","json":"/paper/2311.15930/citation-record.json","paper":"/paper/2311.15930"},"outbound":[],"paper":{"arxiv_id":"2311.15930","last_updated":"2023-11-27T15:38:17Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-13T05:16:56.390928Z","submitted_at":"2023-11-27T15:38:17Z","title":"WorldSense: A Synthetic Benchmark for Grounded Reasoning in Large Language Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"thesis":"As of 14 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 6 inbound Pith citation observations for arXiv:2311.15930."}