{"as_of":"2026-08-19T09:55:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:421fb0914efe6d86105dbb88c143568ef7fa9548034d5fd019a2ff0851da6af3","coverage":[{"denominator":65,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":65,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:56:34.350227Z","state":"measured"},{"denominator":66,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":66,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-19T06:32:44.657259+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-10T15:40:18.133637Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-11T10:06:01.278099Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"cited_work":{"arxiv_id":"2506.02058","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.02058","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Hui Lin and Jeff Bilmes","venue":null,"work_id":"0859602c-7286-44ce-9112-323289a693ca","year":2011},"citing_paper":{"arxiv_id":"2604.12015","last_updated":"2026-04-13T20:00:41Z","snapshot_observed_at":"2026-07-06T23:00:17.634307Z","submitted_at":"2026-04-13T20:00:41Z","title":"UCS: Estimating Unseen Coverage for Improved In-Context Learning","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-10T15:40:18.133637Z"},"links":{"cited_paper":"/paper/2506.02058","citing_paper":"/paper/2604.12015"},"observation_digest":"sha256:b81878c092b35163a6326103ed9cbcf425909de736ae68e6f77ce514c2a9f02b","observation_id":"4f39711d-05df-4367-bdf0-2987ca8ae6db","resolution":{"observed_at":"2026-05-11T10:06:01.283344Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.02058/citation-record","integrity":"/paper/2506.02058/integrity","json":"/paper/2506.02058/citation-record.json","paper":"/paper/2506.02058"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-17T09:58:46.058102Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-07T11:56:29.406168Z","title":"GPT-4 technical report","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:29.406168Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:39b717c19880eb752a432f4c67d29ecbaef6d5df05db92b9548a9f18c43f0c9b","observation_id":"72ab1322-29cf-4754-a082-ab21a584d42e","resolution":{"observed_at":"2026-08-07T11:56:29.406168Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.14316","last_updated":"2024-07-16T10:22:51Z","snapshot_observed_at":"2026-08-16T14:57:46.371369Z","submitted_at":"2023-09-25T17:37:20Z","title":"Physics of Language Models: Part 3.1, Knowledge Storage and Extraction","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.14316","snapshot_observed_at":"2026-08-07T11:56:29.482658Z","title":"Physics of language models: Part 3.1, knowledge storage and extraction","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:29.482658Z"},"links":{"cited_paper":"/paper/2309.14316","citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:4b0b339c7359b8f35dbe711f95a9e3f0fdfd8c726efcaf2d4065c35c76e47ad3","observation_id":"6856b69a-28b4-4b10-85a5-ef242ebea546","resolution":{"observed_at":"2026-08-07T11:56:29.482658Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:35.142945Z","title":"Claude3.7 Sonnetsystemcard","venue":null,"work_id":"4986b376-3270-40ff-8e56-ed50d3981cce","year":2025},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:29.576094Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:3d5f0f6e5a30f1abd2db35b16f6fbbc0d52545fbd05e53ea6a90e9dac9438367","observation_id":"483df34c-5857-4588-a5f8-e33033e45f75","resolution":{"observed_at":"2026-08-07T11:56:35.145924Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2108.07258","last_updated":"2022-07-12T23:45:14Z","snapshot_observed_at":"2026-08-02T09:20:40.804790Z","submitted_at":"2021-08-16T17:50:08Z","title":"On the Opportunities and Risks of Foundation Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2108.07258","snapshot_observed_at":"2026-08-07T11:56:29.637691Z","title":"On the opportunities and risks of foundation models.arXiv preprint arXiv:2108.07258, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:29.637691Z"},"links":{"cited_paper":"/paper/2108.07258","citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:fd6a24c7215647374039c543da458ab1c0bf228d2feb29005cdb1b73d864a421","observation_id":"310dca0d-c683-4f36-8c64-c3e4099af996","resolution":{"observed_at":"2026-08-07T11:56:29.637691Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:29.701569Z","title":"Eight things to know about large language models.Critical AI, 2(2), 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:29.701569Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:f7c147d847e23b3d783284f558134375303de97209d78931740f616b5452dcd3","observation_id":"688d1427-ca52-42e9-8eeb-76a76e84da90","resolution":{"observed_at":"2026-08-07T11:56:29.701569Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:29.768209Z","title":"Language models are few-shot learners","venue":null,"work_id":null,"year":1901},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:29.768209Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:fa31972b8db648dd19a0d68789cc2da5dd1da0cf9d39a7211dbbd483c6f9932c","observation_id":"52c15b9c-a07c-4c0d-8f30-eec444771e7d","resolution":{"observed_at":"2026-08-07T11:56:29.768209Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.12712","last_updated":"2023-04-13T20:41:31Z","snapshot_observed_at":"2026-08-18T22:17:47.865607Z","submitted_at":"2023-03-22T16:51:28Z","title":"Sparks of Artificial General Intelligence: Early experiments with GPT-4","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.12712","snapshot_observed_at":"2026-08-07T11:56:29.866099Z","title":"Sparks of artificial general intelligence: Early experiments with GPT-4","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:29.866099Z"},"links":{"cited_paper":"/paper/2303.12712","citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:63cc3b32ebf9525679aad4df3a5b1a2fd615b71eb300cdbe4308cd1450708bab","observation_id":"eb8c60f9-0d8a-4f75-b3b3-1ac1710b9d85","resolution":{"observed_at":"2026-08-07T11:56:29.866099Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:35.121150Z","title":"Quantifying memorization across neural language models","venue":null,"work_id":"0859d098-3013-42e6-8e1f-16d8b846b595","year":2023},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:29.927015Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:7cfa995373dd09fe8eeec5624e01d5b9a354b85a487e50a8860ef034d9768c6e","observation_id":"c7bf8963-f614-4700-8a85-8fd0267fb6ab","resolution":{"observed_at":"2026-08-07T11:56:35.124339Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:35.111193Z","title":"How do large language models acquire factual knowledge during pretraining? In Neural Information Processing Systems, 2024","venue":null,"work_id":"d2642c17-ee77-4726-a42c-b5bc3a9aaa04","year":2024},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:30.009946Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:b5d08cc08b1714770ea7acb59eee7cde7353313211dc0a5f330040a117533fc0","observation_id":"acbe91c0-5b91-48fb-a285-504003286703","resolution":{"observed_at":"2026-08-07T11:56:35.114386Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:30.089145Z","title":"A survey on evaluation of large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:30.089145Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:f5e255e64a555090676c7f7c483bf981b278ba3911230557cbcdbc77a0e13bc1","observation_id":"3dcb6d7e-dcd2-4478-b92e-73d7c8488979","resolution":{"observed_at":"2026-08-07T11:56:30.089145Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:35.096235Z","title":"Estimating the number of species in a stochastic abundance model","venue":null,"work_id":"05029abf-5057-4159-a4f5-b83d00710194","year":2002},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:30.173634Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:7f45336d678cbbbb11626dde0d41eac509ef39cd876029e1848e98196a59276b","observation_id":"c6a89fa4-efc2-4f5e-abc6-b7a8f00ec216","resolution":{"observed_at":"2026-08-07T11:56:35.099450Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:35.087516Z","title":"A new statistical approach for assessing similarity of species composition with incidence and abundance data","venue":null,"work_id":"eb1e70c7-bb04-4f46-b1e6-8a38196e46d9","year":2005},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:30.264147Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:5ecd3f6303790a9b51c194ed032b142a38b0aaa12c11dfa8d1c18a140c22b78c","observation_id":"c10ddbc8-4a61-4e07-a370-281b9cbf1b2d","resolution":{"observed_at":"2026-08-07T11:56:35.090536Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:35.078815Z","title":"Gonzalez, and Ion Stoica","venue":null,"work_id":"a65a46ad-b6d4-4500-9363-e3e8eb536ed2","year":2024},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:30.373879Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:9c4cb24b7725a9556691146a6e823c9cd655291648ddda569876cf2053b0413b","observation_id":"f6da8e46-f8f9-4d7d-9a97-54c0cf7f2fed","resolution":{"observed_at":"2026-08-07T11:56:35.081757Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-14T02:43:01.480086Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-07T11:56:30.466711Z","title":"Training verifiers to solve math word problems.arXiv preprint arXiv:2110.14168, 2021","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:30.466711Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:472110e551f16485ab8cbf12dde489e87181abe995ab34462fa2b25bee11665d","observation_id":"b03eeb4f-30c9-42f9-96c8-d6df7e737740","resolution":{"observed_at":"2026-08-07T11:56:30.466711Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:30.563192Z","title":"Forgetwhat you know about LLMs evaluations—LLMs are like a chameleon.arXiv preprint arXiv:2502.07445, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:30.563192Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:2ec8466bc07276363b06b20d0df8f8e9ead2db5c38489a4b8c5984a28b2157de","observation_id":"378f9079-eb0f-42d9-8c91-8c2c76031828","resolution":{"observed_at":"2026-08-07T11:56:30.563192Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:30.691748Z","title":"Springer US, Boston, MA, 2009","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:30.691748Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:3a7eeb5a9e08744aee84afd3c6dbf6c8a045338d75a5f68ca7039d5b03c6f252","observation_id":"48ea9d4c-b2af-4788-8e7a-be70489ca6f4","resolution":{"observed_at":"2026-08-07T11:56:30.691748Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:35.069730Z","title":"CURIE: Evaluating LLMs on multitask scientific long-context understanding and reasoning","venue":null,"work_id":"575507b8-5b96-4c80-8d07-f1127dfc53bc","year":2025},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:30.800214Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:d0fe93adc2029002b0efd8a8625be45411ab756ea4aded145c1b4ae4a87aed46","observation_id":"1127a67f-eacf-47b3-8e4f-5f31f4a9b94c","resolution":{"observed_at":"2026-08-07T11:56:35.072802Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:30.902979Z","title":"Data science at the singularity.Harvard Data Science Review, 6(1), 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:30.902979Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:dd59c15405fa7b97cbcb68f42d9475b36d6dbf0cd5480f8249b19bf4ebc95a88","observation_id":"3dd36863-3d46-41a1-b651-f92f6d11ca24","resolution":{"observed_at":"2026-08-07T11:56:30.902979Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:35.054297Z","title":"Estimating the number of unseen species: How many words did Shakespeare know?Biometrika, 63(3):435–447, 1976","venue":null,"work_id":"5fe45319-3409-4a21-b0f0-69ecb9371303","year":1976},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:30.995955Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:8839fb9498b7c6816ea10cd28a26bd4e319ecc8cdf8d7ccebdca810ff55a1241","observation_id":"b88b454a-e470-4841-8662-022324c9d8d5","resolution":{"observed_at":"2026-08-07T11:56:35.057578Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:35.044264Z","title":"Near-optimal estimation of the unseen under regularly varying tail populations.Bernoulli, 29(4):3423–3442, 2023","venue":null,"work_id":"a359c372-e884-4183-b0f8-7b4d60ea6732","year":2023},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:31.108281Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:4c6f71b77cab9f9f6993c1df1c06db8b47a7229e1f6bb83472e2b3aa86ae4e66","observation_id":"fec3978d-2ff7-424c-a75c-b4eb4fbd38ac","resolution":{"observed_at":"2026-08-07T11:56:35.047540Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:31.201554Z","title":"Good-turing frequency estimation without tears.Journal of quantitative linguistics, 2(3):217–237, 1995","venue":null,"work_id":null,"year":1995},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:31.201554Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:19e645fc2cf0cea3816aa48c3083301e5e0afeb410562c343a29c90555ebd3a5","observation_id":"ca5772c1-a696-4352-a239-2cbaa2b9bfed","resolution":{"observed_at":"2026-08-07T11:56:31.201554Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:35.027942Z","title":"The population frequencies of species and the estimation of population parameters","venue":null,"work_id":"894c9f78-0683-42fc-ab0e-c7417f91c73c","year":1953},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:31.316125Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:49070c85f4bc35d4e1c5069c90a199fe8b3aba05763961c7457e28234d37a194","observation_id":"2f94198c-03f9-46b7-83f1-1ac2be5258f3","resolution":{"observed_at":"2026-08-07T11:56:35.031401Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:35.018086Z","title":"Estimating knowledge in large language models without generating a single token","venue":null,"work_id":"4d7b8168-7167-480e-9ca9-ece28385c62b","year":2024},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:31.399782Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:ab7f55b230daeecb54a9a617d088b1be9e0433c4313a627cc55db93dfe2950c9","observation_id":"f11c90b2-fe19-45a4-9762-1131b0048e9b","resolution":{"observed_at":"2026-08-07T11:56:35.021476Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-07T11:56:31.486703Z","title":"The Llama 3 herd of models.arXiv preprint arXiv:2407.21783, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:31.486703Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:ab240036e3ee85390583960e156e806ebbf9bdc1dd0fe63cc04fc07b7ad3f838","observation_id":"f17a9559-781b-4bc9-8727-ba3f6e3ed67c","resolution":{"observed_at":"2026-08-07T11:56:31.486703Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:35.007663Z","title":"Optimal prediction of the number of unseen species with multiplicity","venue":null,"work_id":"7244dee8-7e01-4fbd-af8a-104e9ff0944a","year":2020},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:31.589683Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:6662a1ed8089ac09bc09c09eb07d7d8a07a45e2845dc3930da28d6ae3f16bc6f","observation_id":"aae552b3-fbaa-4cbf-9c85-4b0dfad527ba","resolution":{"observed_at":"2026-08-07T11:56:35.011026Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:31.701733Z","title":"Measuring massive multitask language understanding","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:31.701733Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:7340ba6e2f97ae3275d1558f8c702882d12974f72268dcb5889f3d5b94f46e22","observation_id":"e1b4f6e4-a378-4ed5-83a4-01e165bbcbb1","resolution":{"observed_at":"2026-08-07T11:56:31.701733Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:31.806850Z","title":"The curious case of neural text degeneration","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:31.806850Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:faf08f72bcb0472dfa3c3e4e037be2944df1b60118e7f7987014ce556aa155f4","observation_id":"2d78d870-0ef7-45a6-8018-4b91dc929097","resolution":{"observed_at":"2026-08-07T11:56:31.806850Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-08-15T14:02:47.366139Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-07T11:56:31.905024Z","title":"GPT-4o system card.arXiv preprint arXiv:2410.21276, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:31.905024Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:7ca963adcae4ba5ec3d2a6b2470b5351b9bd1951ec7abd87491731f2bc0a4c75","observation_id":"3369593b-ec38-4dc2-93f2-35b1cbc63d97","resolution":{"observed_at":"2026-08-07T11:56:31.905024Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06825","last_updated":"2023-10-10T17:54:58Z","snapshot_observed_at":"2026-08-17T20:30:34.016254Z","submitted_at":"2023-10-10T17:54:58Z","title":"Mistral 7B","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06825","snapshot_observed_at":"2026-08-07T11:56:32.057825Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:32.057825Z"},"links":{"cited_paper":"/paper/2310.06825","citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:80433c46fa9134bded48995db1751f33cdcea106a3081733a49195d90e7cd216","observation_id":"8b7ece32-5645-4610-a2fb-3ebc3a7c75dc","resolution":{"observed_at":"2026-08-07T11:56:32.057825Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.985228Z","title":"Calibrated language models must hallucinate","venue":null,"work_id":"06dbedf8-c8a7-4777-9773-a0be618fdbd5","year":2024},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:32.156600Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:429be7504258afdae13eedafa3a38eca68886a39e42b0e3b4d88d0b4039c01c4","observation_id":"e6655ea8-530c-4da3-ae41-b5589d52714f","resolution":{"observed_at":"2026-08-07T11:56:34.989799Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2503.11256","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.649439Z","title":"Line of duty: Evaluating LLM self-knowledge via consistency in feasibility boundaries.arXiv preprint arXiv:2503.11256, 2025","venue":null,"work_id":"125431ef-f3b6-4637-b37a-a9759d993a91","year":2025},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:32.274696Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:429162656fcc703ab769afc0e7baf9b1028c306c6d37801bc078b765b7bb0285","observation_id":"b677d7cb-2225-45af-b630-69fe5edaf29d","resolution":{"observed_at":"2026-08-07T11:56:34.654800Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.975963Z","title":"Too many AIs.https://dev.to/leeaao/too-many-ais-24nb, 2025","venue":null,"work_id":"e9ab2ae4-5461-415b-9ef3-563dde05c64a","year":2025},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:32.398334Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:07dbc39fee8df12157675cb9fb811595132c8ea1473ff0629a5668ffd15defb5","observation_id":"aec70103-f245-46c7-a6cd-500493c24cc3","resolution":{"observed_at":"2026-08-07T11:56:34.979147Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.966892Z","title":"BioASQ-QA: A manually curated corpus for biomedical question answering.Scientific Data, 10 (1):170, 2023","venue":null,"work_id":"e1d3bf7f-4484-47ba-92ba-bebd02a6c4e6","year":2023},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:32.490401Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:d1ce27a96d56e937c3a50f0ace064e2a0068b1bcc450dc1529e52c7fd2272df2","observation_id":"57c732cf-9d26-4733-8cec-29c15d6a59d4","resolution":{"observed_at":"2026-08-07T11:56:34.969869Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.957284Z","title":"How pre-trained language models capture factual knowledge? A causal-inspired analysis","venue":null,"work_id":"33798cfe-7523-4d95-b49d-b47b18659c48","year":2022},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:32.562356Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:29126381a02c8d0e19b89fabced1419dc06183ced485d722269bab3c78254fc2","observation_id":"1fa5d269-dc34-47e8-bc34-dcdc8e918390","resolution":{"observed_at":"2026-08-07T11:56:34.960897Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.948560Z","title":"ROUGE: A package for automatic evaluation of summaries","venue":null,"work_id":"bafe8b6d-015d-4f0f-9ac3-535d244dffd1","year":2004},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:32.665033Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:7ad6316832b25ab7d004122d5ce0b43d8084383dd1dd20c2876fa082b2da2839","observation_id":"95f36e0b-4d3d-4373-b917-47032910befc","resolution":{"observed_at":"2026-08-07T11:56:34.951469Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-08-18T18:18:37.449517Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-08-07T11:56:32.742702Z","title":"Deepseek-V3 technical report.arXiv preprint arXiv:2412.19437, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:32.742702Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:cbe816dec82d8d3aed32549049c842fd830a6556c9bface2c11fe27fe3306f66","observation_id":"a9b5a12a-3641-4fcb-9030-656528f8c9ac","resolution":{"observed_at":"2026-08-07T11:56:32.742702Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.939704Z","title":"Feder Cooper, Daphne Ippolito, Christopher A","venue":null,"work_id":"9f6a97fa-8277-4052-84e6-860e04f4aea5","year":2025},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:32.831463Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:05acc9abf4161512e36f116c667806a90fc732d2975ad65d93495e010d192a70","observation_id":"0ab7b408-c554-4834-8f88-2094453ab49a","resolution":{"observed_at":"2026-08-07T11:56:34.942557Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.931128Z","title":"ChatGPT-3.5-turbo","venue":null,"work_id":"1dd93daa-203c-4bb3-b17c-1f4d5b0f8759","year":2023},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:32.939632Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:ae08a8b8424d72fd2ae599076b59a51ee67a98071fce96006cf73f2ffb3cfdbb","observation_id":"9bdadb55-1361-4762-aa38-54a9fcac97a8","resolution":{"observed_at":"2026-08-07T11:56:34.933728Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.922523Z","title":"Competitive distribution estimation: Why is Good-Turing good","venue":null,"work_id":"2bdb1e6c-208a-4979-a79b-1f50d7f9255a","year":2015},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:33.057660Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:e7839410d0824ef3c0a35f45642073a3b0d8e880c1114522c2c64da793e37ef3","observation_id":"85ce83e6-eff3-4976-9842-f6f8c64f84f4","resolution":{"observed_at":"2026-08-07T11:56:34.925398Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.913965Z","title":"Always Good Turing: Asymptotically optimal probability estimation.Science, 302(5644):427–431, 2003","venue":null,"work_id":"fe13aecf-eaa6-48ca-8575-2a5a6adf4d11","year":2003},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:33.170593Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:659a06ee1b2e61a79b22652af3423194d3ed52d77a2c802753752b1dbb9e20b6","observation_id":"f17f5189-d0c5-46f6-8727-8bcb583d3a86","resolution":{"observed_at":"2026-08-07T11:56:34.916852Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.904251Z","title":"Optimal prediction of the number of unseen species","venue":null,"work_id":"cc077b99-1101-4bc5-8944-a83f176dc0e1","year":2016},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:33.261347Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:dc52c6e5d9689a2485c3e0958cd6c5734abd1390da1283d6415ac32ebc75af47","observation_id":"f2c6e254-e81a-4a3b-8fbc-06a456d11fd1","resolution":{"observed_at":"2026-08-07T11:56:34.907642Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.894398Z","title":"BLEU: A method for automatic evaluation of machine translation","venue":null,"work_id":"6087d47f-162c-48a4-b971-4863957675cd","year":2002},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:33.392631Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:41c87383135551a144dd9930aaf5978dda895c5aaf94ff9cbb19d192b0d287eb","observation_id":"ce2a9fe4-7ca7-4a92-98d4-b5a1a592c3c0","resolution":{"observed_at":"2026-08-07T11:56:34.897789Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.885466Z","title":null,"venue":null,"work_id":"127a3e52-ee18-43e8-968b-c38872c2d32f","year":2019},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:33.542099Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:8afb740fe63a621584a338c0136bbdb33e3ad0ec13bfbdcd5da6e43b8ec069b8","observation_id":"0faef811-19d8-4ae0-9219-45c3b109bcb2","resolution":{"observed_at":"2026-08-07T11:56:34.888538Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14249","last_updated":"2026-02-20T04:23:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-24T05:27:46Z","title":"Humanity's Last Exam","version":10},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.14249","snapshot_observed_at":"2026-08-07T11:56:33.693512Z","title":"Humanity’s last exam.arXiv preprint arXiv:2501.14249, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:33.693512Z"},"links":{"cited_paper":"/paper/2501.14249","citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:27913b5714e6c2138df3eb400d34726b4b1c19fabb971c4cbd67d9f44b6098d7","observation_id":"68e09110-55f4-421b-9086-8ebd96634a3c","resolution":{"observed_at":"2026-08-07T11:56:33.693512Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:33.840301Z","title":"Do large language models know how much they know?arXiv preprint arXiv:2502.19573, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:33.840301Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:3f07678f4562a9f1fd48610410d51606fcd820aec9d3cb1ea317f75fdf5aba80","observation_id":"900d1761-f576-45ef-b239-387891575e95","resolution":{"observed_at":"2026-08-07T11:56:33.840301Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.876505Z","title":"AI and the everything in the whole wide world benchmark","venue":null,"work_id":"b1c093cc-21fa-4141-a060-36569e5bcad2","year":2021},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:33.927131Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:85353765044e5b8927c8b41335d32adb903a4160bd9521434f2beb2afafc9715","observation_id":"96fb3030-8cd4-4bf9-90ff-ba9b27161418","resolution":{"observed_at":"2026-08-07T11:56:34.879529Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.867164Z","title":"NLP evaluation in trouble: On the need to measure LLM data contamination for each benchmark","venue":null,"work_id":"15442787-f265-4264-a901-32f861418f89","year":2023},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:33.972092Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:02b36561a78febfdde12d752145068238a676680f6702f028f6f233e672f0222","observation_id":"4b1c0bfa-8e40-4db1-82b9-0aa6c63ca32c","resolution":{"observed_at":"2026-08-07T11:56:34.870380Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.13507","last_updated":"2025-03-13T19:35:40Z","snapshot_observed_at":"2026-08-16T12:50:14.319584Z","submitted_at":"2025-03-13T19:35:40Z","title":"NeurIPS 2023 LLM Efficiency Fine-tuning Competition","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.13507","snapshot_observed_at":"2026-08-07T11:56:34.073212Z","title":"NeurIPS 2023 LLM efficiency fine-tuning competition","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.073212Z"},"links":{"cited_paper":"/paper/2503.13507","citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:86fdb585056f01f9935951ec7e1568152b6eab594ea5503ea2be77493384917d","observation_id":"af1bb3c2-8c34-48f4-ba7e-f3450f84ede4","resolution":{"observed_at":"2026-08-07T11:56:34.073212Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1093/nar/gkab920","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"Human disease ontology 2022 update","venue":"Nucleic Acids Research","work_id":"ab4ca00b-3517-4ae3-a1f3-623468d7cf85","year":2022},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.173184Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:de422912b4e78fe61a383726b8e6a15d4e937445c0a4c648c7c25c40274b48c6","observation_id":"fadec12e-09ec-48bd-bdad-4678fb372d40","resolution":{"observed_at":"2026-08-07T11:56:34.393035Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.857550Z","title":"Auto- Prompt: Eliciting knowledge from language models with automatically generated prompts","venue":null,"work_id":"523624f5-200d-4fbb-babb-8ee6d14d741b","year":2020},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.298457Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:ce4ba14abee0385cda712b67fadf49d99ee24d59ac0f22c6e4862210c7beb7a3","observation_id":"bfa37a32-4556-4365-bde5-612215903355","resolution":{"observed_at":"2026-08-07T11:56:34.860670Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.304140Z","title":"Welcome to the era of experience.Google AI, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.304140Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:3e8453a247268c086849e71e90e129331e9cf0337062a4a4d34e5a687088d7d7","observation_id":"4278205c-6193-4b12-809a-f2a1b2779fd5","resolution":{"observed_at":"2026-08-07T11:56:34.304140Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20879","last_updated":"2025-05-12T16:33:58Z","snapshot_observed_at":"2026-08-17T00:43:37.683523Z","submitted_at":"2025-04-29T15:48:49Z","title":"The Leaderboard Illusion","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20879","snapshot_observed_at":"2026-08-07T11:56:34.307356Z","title":"Smith, Beyza Ermis, Marzieh Fadaee, and Sara Hooker","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.307356Z"},"links":{"cited_paper":"/paper/2504.20879","citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:4de3f78175662bb5fb6114fe2e2b741ba5f80c78abd526140a4b55f37e461654","observation_id":"cb6f8d71-aa20-4602-a8c0-b521e391c82c","resolution":{"observed_at":"2026-08-07T11:56:34.307356Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.311072Z","title":"Large language models encode clinical knowledge.Nature, 620(7972):172–180, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.311072Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:89bba5b60f4c572532644704524dda4e087b2bf0167b7f23065c173e3f13d16d","observation_id":"0714b062-436d-42fa-8830-9680c3342321","resolution":{"observed_at":"2026-08-07T11:56:34.311072Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.314529Z","title":"The bitter lesson.Incomplete Ideas (blog), 13(1):38, 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.314529Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:dc44ade7f4c33578119338744dbca62e00c683c41d02e1f1a36fd91a0dfffdbd","observation_id":"63154cd9-7896-4b3b-bc9f-231e38059ce7","resolution":{"observed_at":"2026-08-07T11:56:34.314529Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2211.09085","last_updated":"2022-11-16T18:06:33Z","snapshot_observed_at":"2026-08-17T08:35:32.078396Z","submitted_at":"2022-11-16T18:06:33Z","title":"Galactica: A Large Language Model for Science","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2211.09085","snapshot_observed_at":"2026-08-07T11:56:34.317573Z","title":"Galactica: A large language model for science.arXiv preprint arXiv:2211.09085, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.317573Z"},"links":{"cited_paper":"/paper/2211.09085","citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:04fc35a9d29f4769135fedb9afb5b05ccf611c5f1dd8660571c01e18067b36e3","observation_id":"cdf82a61-b4c1-4319-99ba-e27ea41fb45c","resolution":{"observed_at":"2026-08-07T11:56:34.317573Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05530","last_updated":"2024-12-16T17:39:39Z","snapshot_observed_at":"2026-08-14T18:15:53.516440Z","submitted_at":"2024-03-08T18:54:20Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05530","snapshot_observed_at":"2026-08-07T11:56:34.320882Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context.arXiv preprint arXiv:2403.05530, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.320882Z"},"links":{"cited_paper":"/paper/2403.05530","citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:58e26aace53cebb227868612ea99872c6e4c353154571d7cba22b3e9433f53fa","observation_id":"bf256b5f-6e75-449f-bc4e-52fb3deaba4d","resolution":{"observed_at":"2026-08-07T11:56:34.320882Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.5749/j.ctttv","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.377703Z","title":null,"venue":null,"work_id":"38535c0f-a85b-4348-8dd5-b62ad85b188a","year":1957},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.324022Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:8a1855c53cd3c2f9879f0e5d0ff6568ee9fde52d3554d17bf83876ad9ad9bae4","observation_id":"18c493fd-f9b1-4e8b-9689-78de7603398e","resolution":{"observed_at":"2026-08-07T11:56:34.382001Z","resolver_source":"doi_truncated","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11967","last_updated":"2020-10-22T18:01:56Z","snapshot_observed_at":"2026-08-16T19:11:21.654742Z","submitted_at":"2020-10-22T18:01:56Z","title":"Language Models are Open Knowledge Graphs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11967","snapshot_observed_at":"2026-08-07T11:56:34.327247Z","title":"Language models are open knowledge graphs","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.327247Z"},"links":{"cited_paper":"/paper/2010.11967","citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:8264b99eec7a3db85078db6fe8d35497406101af0b44493a1607263446f5487f","observation_id":"64e7e780-3939-43a2-953a-1324e6ab1c40","resolution":{"observed_at":"2026-08-07T11:56:34.327247Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.01623","last_updated":"2024-01-25T13:10:15Z","snapshot_observed_at":"2026-08-16T14:30:09.258598Z","submitted_at":"2024-01-03T08:49:12Z","title":"Can AI Be as Creative as Humans?","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.01623","snapshot_observed_at":"2026-08-07T11:56:34.330717Z","title":"Can AI be as creative as humans? arXiv preprint arXiv:2401.01623, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.330717Z"},"links":{"cited_paper":"/paper/2401.01623","citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:11e2909f5161e675fe5c8aa45e688f8311904cd477fb39b24c27bbb34f99bc1c","observation_id":"60d4733b-45b1-4bef-880f-000b5a15117c","resolution":{"observed_at":"2026-08-07T11:56:34.330717Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.828336Z","title":"Chi, Quoc V","venue":null,"work_id":"1ac78e23-8329-49c2-8f3c-e35c03245bf7","year":2022},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.334226Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:d28016921d2ed7dd029461a026b5ad8b2063f13dd59d5cd359ab33cca4499633","observation_id":"5d323f19-07f3-41a9-9f95-676aaa7aae99","resolution":{"observed_at":"2026-08-07T11:56:34.830980Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.819081Z","title":"Estimating the probabilities of rare outputs in language models","venue":null,"work_id":"21b0ccc3-37fe-4dc1-ad1a-9a88217c4aa4","year":2025},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.337670Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:069f0832b26caca420304779be3bc23af611c047b638b9762115bac4cfcbfaaf","observation_id":"40608028-8344-4c9d-bbfe-42717f898ea3","resolution":{"observed_at":"2026-08-07T11:56:34.822545Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-08-17T18:50:07.059564Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-07T11:56:34.340659Z","title":"Qwen2.5 technical report.arXiv preprint arXiv:2412.15115, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.340659Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:35d0df083b3b9d49a0244cb3e0d30bb48d21fe2d7170ee14e0988f0f29a6f460","observation_id":"8f4f37ea-0c4b-4110-a534-f91081bc8a5e","resolution":{"observed_at":"2026-08-07T11:56:34.340659Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.809970Z","title":"A careful examination of large language model performance on grade school arithmetic","venue":null,"work_id":"98fdccfb-627e-4efb-a742-3b14546c0118","year":2024},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.343378Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:881c41acb4e65f78da555cc1698685ac808d3b5de3b5d2cee5aa0b100f16d746","observation_id":"893d72c8-1572-424b-9370-94f79a7c44b2","resolution":{"observed_at":"2026-08-07T11:56:34.812963Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.06779","last_updated":"2024-07-09T11:48:49Z","snapshot_observed_at":"2026-08-16T13:35:49.287572Z","submitted_at":"2024-07-09T11:48:49Z","title":"Using Pretrained Large Language Model with Prompt Engineering to Answer Biomedical Questions","version":1},"cited_work":{"arxiv_id":"2407.06779","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.06779","snapshot_observed_at":"2026-08-07T11:56:34.409879Z","title":"Using Pretrained Large Language Model with Prompt Engineering to Answer Biomedical Questions","venue":"cs.CL","work_id":"c23055ea-e176-43d8-9054-5fbdbd7a7c57","year":2024},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.346706Z"},"links":{"cited_paper":"/paper/2407.06779","citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:da1b58645d746626d9e25d9d6e6383388942e1d31a7594aabc5b635b5ef3aaea","observation_id":"88fcaddc-d668-4a60-87fd-a06a821f89a5","resolution":{"observed_at":"2026-08-07T11:56:34.413661Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.800654Z","title":"Test your knowledge of mathematical theorems by listing 20 theorem names, separated by commas","venue":null,"work_id":"b950616e-dba1-481c-9670-2c9249bda57a","year":2024},"citing_paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.350227Z"},"links":{"citing_paper":"/paper/2506.02058"},"observation_digest":"sha256:326797ef9509fad51bc21b26e5aaf897bc9c039849e03b0e0573756263232dc5","observation_id":"29d200ca-a50b-460f-8047-70ddb01d209f","resolution":{"observed_at":"2026-08-07T11:56:34.803679Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.02058","last_updated":"2025-06-01T15:32:44Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-17T17:06:21.222216Z","submitted_at":"2025-06-01T15:32:44Z","title":"Evaluating the Unseen Capabilities: How Many Theorems Do LLMs Know?"},"reference_resolution":{"displayed":65,"state_counts":{"malformed_identifier":2,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":31,"verified_exact":3,"verified_fuzzy":29},"total_outbound_references":65},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"thesis":"As of 19 August 2026, this Paper Citation Record lists 65 of 65 outbound references and 1 inbound Pith citation observation for arXiv:2506.02058."}