{"as_of":"2026-08-12T06:32:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:61cd50dd4fa7ef42aaf3e7d5ef371bbb2aa50a944c1409822c7ffdf8e8509fb9","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":18,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":18,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":18,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":18,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T00:04:55.376776Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":5,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2310.17623","last_updated":"2023-11-24T01:45:16Z","snapshot_observed_at":"2026-07-06T16:39:08.850847Z","submitted_at":"2023-10-26T17:43:13Z","title":"Proving Test Set Contamination in Black Box Language Models","version":2},"cited_work":{"arxiv_id":"2310.17623","doi":"10.48550/arxiv.2310.17623","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17623","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatterji, Faisal Ladhak, and Tatsunori Hashimoto","venue":"arXiv (Cornell University)","work_id":"cf915c30-398f-4fa5-ad6c-b5a693575904","year":2023},"citing_paper":{"arxiv_id":"2403.03952","last_updated":"2026-04-20T06:05:54Z","snapshot_observed_at":"2026-08-03T00:56:47.399756Z","submitted_at":"2024-03-06T18:56:36Z","title":"Bridging Language and Items for Retrieval and Recommendation: Benchmarking LLMs as Semantic Encoders","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-05-24T03:03:51.053556Z"},"links":{"cited_paper":"/paper/2310.17623","citing_paper":"/paper/2403.03952"},"observation_digest":"sha256:0eb8d75b057784c1cc6eabc119229dd936bf6d2dda1fc78d82da7b7941102aa9","observation_id":"aac2b93c-33af-42fd-9be3-c62381b806c8","resolution":{"observed_at":"2026-05-24T03:05:56.957841Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17623","last_updated":"2023-11-24T01:45:16Z","snapshot_observed_at":"2026-07-06T16:39:08.850847Z","submitted_at":"2023-10-26T17:43:13Z","title":"Proving Test Set Contamination in Black Box Language Models","version":2},"cited_work":{"arxiv_id":"2310.17623","doi":"10.48550/arxiv.2310.17623","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17623","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatterji, Faisal Ladhak, and Tatsunori Hashimoto","venue":"arXiv (Cornell University)","work_id":"cf915c30-398f-4fa5-ad6c-b5a693575904","year":2023},"citing_paper":{"arxiv_id":"2406.19314","last_updated":"2025-04-18T19:36:00Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-27T16:47:42Z","title":"LiveBench: A Challenging, Contamination-Limited LLM Benchmark","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-15T04:48:26.303240Z"},"links":{"cited_paper":"/paper/2310.17623","citing_paper":"/paper/2406.19314"},"observation_digest":"sha256:2eebdaeee571c07817a37fc187c5832da8f1e009ec072a2582d638b91a2022f7","observation_id":"9cfa3985-199b-47ea-85c2-b13326b0d4d0","resolution":{"observed_at":"2026-05-15T04:48:26.469976Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17623","last_updated":"2023-11-24T01:45:16Z","snapshot_observed_at":"2026-07-06T16:39:08.850847Z","submitted_at":"2023-10-26T17:43:13Z","title":"Proving Test Set Contamination in Black Box Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.17623","snapshot_observed_at":"2026-08-12T00:04:55.376776Z","title":"Hashimoto","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.03597","last_updated":"2024-12-02T20:49:21Z","snapshot_observed_at":"2026-08-11T23:56:38.089193Z","submitted_at":"2024-12-02T20:49:21Z","title":"The Vulnerability of Language Model Benchmarks: Do They Accurately Reflect True LLM Performance?","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-12T00:04:55.376776Z"},"links":{"cited_paper":"/paper/2310.17623","citing_paper":"/paper/2412.03597"},"observation_digest":"sha256:c665a27fc216c53b2872a3fa1aa767eec7ca35a6969e94e9aca4425c0cd0e2bb","observation_id":"46758629-71e8-4b08-a4df-f06569258c15","resolution":{"observed_at":"2026-08-12T00:04:55.376776Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17623","last_updated":"2023-11-24T01:45:16Z","snapshot_observed_at":"2026-07-06T16:39:08.850847Z","submitted_at":"2023-10-26T17:43:13Z","title":"Proving Test Set Contamination in Black Box Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.17623","snapshot_observed_at":"2026-08-11T20:30:10.273799Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.05734","last_updated":"2025-08-08T09:27:21Z","snapshot_observed_at":"2026-08-11T20:22:21.290143Z","submitted_at":"2024-12-07T20:09:01Z","title":"LeakAgent: RL-based Red-teaming Agent for LLM Privacy Leakage","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-11T20:30:10.273799Z"},"links":{"cited_paper":"/paper/2310.17623","citing_paper":"/paper/2412.05734"},"observation_digest":"sha256:7625eea5a332dc30ec314e7b38f0976568fc9f73abde8aa1e94fcbad054882b4","observation_id":"09bf5eeb-7e7e-414f-a8c6-e6df7b607cf0","resolution":{"observed_at":"2026-08-11T20:30:10.273799Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17623","last_updated":"2023-11-24T01:45:16Z","snapshot_observed_at":"2026-07-06T16:39:08.850847Z","submitted_at":"2023-10-26T17:43:13Z","title":"Proving Test Set Contamination in Black Box Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.17623","snapshot_observed_at":"2026-08-11T12:58:02.141431Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.13670","last_updated":"2025-05-29T03:19:42Z","snapshot_observed_at":"2026-08-12T04:44:16.092258Z","submitted_at":"2024-12-18T09:53:12Z","title":"AntiLeakBench: Preventing Data Contamination by Automatically Constructing Benchmarks with Updated Real-World Knowledge","version":2},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-11T12:58:02.141431Z"},"links":{"cited_paper":"/paper/2310.17623","citing_paper":"/paper/2412.13670"},"observation_digest":"sha256:6428408f3e948c9b7d8c4913b0f93045a85f511a60a4e935d022286dcdda2d4f","observation_id":"ff05ac5e-4b84-4abe-bf3a-d3bf79d327ac","resolution":{"observed_at":"2026-08-11T12:58:02.141431Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17623","last_updated":"2023-11-24T01:45:16Z","snapshot_observed_at":"2026-07-06T16:39:08.850847Z","submitted_at":"2023-10-26T17:43:13Z","title":"Proving Test Set Contamination in Black Box Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.17623","snapshot_observed_at":"2026-08-09T22:35:45.117286Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.18771","last_updated":"2025-01-30T21:51:18Z","snapshot_observed_at":"2026-08-11T21:32:44.092264Z","submitted_at":"2025-01-30T21:51:18Z","title":"Overestimation in LLM Evaluation: A Controlled Large-Scale Study on Data Contamination's Impact on Machine Translation","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-09T22:35:45.117286Z"},"links":{"cited_paper":"/paper/2310.17623","citing_paper":"/paper/2501.18771"},"observation_digest":"sha256:44c43ecf36b9691bfc63fb0d3692449cf0c57c87b796052e6ea854c1581fac84","observation_id":"c2f3c6fd-1ea6-42c7-93a5-4daaccee9f63","resolution":{"observed_at":"2026-08-09T22:35:45.117286Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17623","last_updated":"2023-11-24T01:45:16Z","snapshot_observed_at":"2026-07-06T16:39:08.850847Z","submitted_at":"2023-10-26T17:43:13Z","title":"Proving Test Set Contamination in Black Box Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.17623","snapshot_observed_at":"2026-08-07T23:44:10.838810Z","title":"Proving test set contamination in black box language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.08794","last_updated":"2025-02-12T21:17:30Z","snapshot_observed_at":"2026-08-10T03:51:48.258660Z","submitted_at":"2025-02-12T21:17:30Z","title":"Spectral Journey: How Transformers Predict the Shortest Path","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T23:44:10.838810Z"},"links":{"cited_paper":"/paper/2310.17623","citing_paper":"/paper/2502.08794"},"observation_digest":"sha256:0dd99bab42165220b78aeb861e54e3d479be1a02077692973778546bc91f15f2","observation_id":"5c32f891-3e89-45c4-b64b-937f3bcb1067","resolution":{"observed_at":"2026-08-07T23:44:10.838810Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17623","last_updated":"2023-11-24T01:45:16Z","snapshot_observed_at":"2026-07-06T16:39:08.850847Z","submitted_at":"2023-10-26T17:43:13Z","title":"Proving Test Set Contamination in Black Box Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.17623","snapshot_observed_at":"2026-08-07T00:16:19.161532Z","title":"Proving test set contamination in black box language models.arXiv preprint arXiv:2310.17623, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.14913","last_updated":"2025-06-17T18:46:45Z","snapshot_observed_at":"2026-08-08T01:57:43.900224Z","submitted_at":"2025-06-17T18:46:45Z","title":"Winter Soldier: Backdooring Language Models at Pre-Training with Indirect Data Poisoning","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T00:16:19.161532Z"},"links":{"cited_paper":"/paper/2310.17623","citing_paper":"/paper/2506.14913"},"observation_digest":"sha256:c9596b59a7ed46f007ecf1c94b1e8a96076d6b83b702b05021329317afbda7bd","observation_id":"1d03e316-85ff-4de6-9ea5-989c62c52be0","resolution":{"observed_at":"2026-08-07T00:16:19.161532Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17623","last_updated":"2023-11-24T01:45:16Z","snapshot_observed_at":"2026-07-06T16:39:08.850847Z","submitted_at":"2023-10-26T17:43:13Z","title":"Proving Test Set Contamination in Black Box Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.17623","snapshot_observed_at":"2026-08-06T15:46:51.772651Z","title":"Proving test set contamination in black box lan- guage models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.15104","last_updated":"2026-06-15T19:35:17Z","snapshot_observed_at":"2026-08-09T15:47:37.792059Z","submitted_at":"2025-07-20T19:57:07Z","title":"AnalogFed: Privacy-Preserving Discovery of Analog Circuits at Scale with Federated Generative AI","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T15:46:51.772651Z"},"links":{"cited_paper":"/paper/2310.17623","citing_paper":"/paper/2507.15104"},"observation_digest":"sha256:17259ad84fa13a49276e62ee4a1a79d14cb7bfc76f879e2f6b619f422ff199c3","observation_id":"1ebef60e-02ce-4974-88e8-e1b3732d9cd9","resolution":{"observed_at":"2026-08-06T15:46:51.772651Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17623","last_updated":"2023-11-24T01:45:16Z","snapshot_observed_at":"2026-07-06T16:39:08.850847Z","submitted_at":"2023-10-26T17:43:13Z","title":"Proving Test Set Contamination in Black Box Language Models","version":2},"cited_work":{"arxiv_id":"2310.17623","doi":"10.48550/arxiv.2310.17623","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17623","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatterji, Faisal Ladhak, and Tatsunori Hashimoto","venue":"arXiv (Cornell University)","work_id":"cf915c30-398f-4fa5-ad6c-b5a693575904","year":2023},"citing_paper":{"arxiv_id":"2604.13371","last_updated":"2026-04-15T00:35:22Z","snapshot_observed_at":"2026-08-08T12:23:50.044941Z","submitted_at":"2026-04-15T00:35:22Z","title":"Empirical Evidence of Complexity-Induced Limits in Large Language Models on Finite Discrete State-Space Problems with Explicit Validity Constraints","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-05-10T14:12:45.438246Z"},"links":{"cited_paper":"/paper/2310.17623","citing_paper":"/paper/2604.13371"},"observation_digest":"sha256:959570d5a55345ebb88a175de6f10d57e601ef329b176e4abf5d0b13f9c1446a","observation_id":"9d5f236f-2fa0-4c09-9b91-3952121237e2","resolution":{"observed_at":"2026-05-10T14:15:29.379804Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17623","last_updated":"2023-11-24T01:45:16Z","snapshot_observed_at":"2026-07-06T16:39:08.850847Z","submitted_at":"2023-10-26T17:43:13Z","title":"Proving Test Set Contamination in Black Box Language Models","version":2},"cited_work":{"arxiv_id":"2310.17623","doi":"10.48550/arxiv.2310.17623","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17623","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatterji, Faisal Ladhak, and Tatsunori Hashimoto","venue":"arXiv (Cornell University)","work_id":"cf915c30-398f-4fa5-ad6c-b5a693575904","year":2023},"citing_paper":{"arxiv_id":"2604.17771","last_updated":"2026-04-20T03:50:21Z","snapshot_observed_at":"2026-08-11T06:02:07.490505Z","submitted_at":"2026-04-20T03:50:21Z","title":"SPENCE: A Syntactic Probe for Detecting Contamination in NL2SQL Benchmarks","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-05-10T04:40:42.747298Z"},"links":{"cited_paper":"/paper/2310.17623","citing_paper":"/paper/2604.17771"},"observation_digest":"sha256:6ad1c16a0cbdaf84e7c9f7ede735f43e0be63313390a7f889e32b075120a21dc","observation_id":"766a0e8b-f878-4ff2-af4c-2d15b870a944","resolution":{"observed_at":"2026-05-10T04:45:21.220860Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17623","last_updated":"2023-11-24T01:45:16Z","snapshot_observed_at":"2026-07-06T16:39:08.850847Z","submitted_at":"2023-10-26T17:43:13Z","title":"Proving Test Set Contamination in Black Box Language Models","version":2},"cited_work":{"arxiv_id":"2310.17623","doi":"10.48550/arxiv.2310.17623","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17623","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatterji, Faisal Ladhak, and Tatsunori Hashimoto","venue":"arXiv (Cornell University)","work_id":"cf915c30-398f-4fa5-ad6c-b5a693575904","year":2023},"citing_paper":{"arxiv_id":"2605.06865","last_updated":"2026-05-07T19:06:35Z","snapshot_observed_at":"2026-07-06T23:19:15.382283Z","submitted_at":"2026-05-07T19:06:35Z","title":"Dataset Watermarking for Closed LLMs with Provable Detection","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-11T00:53:42.185498Z"},"links":{"cited_paper":"/paper/2310.17623","citing_paper":"/paper/2605.06865"},"observation_digest":"sha256:1afa32a4d7c2126f1f9e008756418ba4b3f8e33cd724a598065459a54845cdec","observation_id":"6c932fc7-89e8-471e-9157-f2d3fe42e545","resolution":{"observed_at":"2026-05-11T05:00:57.294124Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17623","last_updated":"2023-11-24T01:45:16Z","snapshot_observed_at":"2026-07-06T16:39:08.850847Z","submitted_at":"2023-10-26T17:43:13Z","title":"Proving Test Set Contamination in Black Box Language Models","version":2},"cited_work":{"arxiv_id":"2310.17623","doi":"10.48550/arxiv.2310.17623","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17623","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatterji, Faisal Ladhak, and Tatsunori Hashimoto","venue":"arXiv (Cornell University)","work_id":"cf915c30-398f-4fa5-ad6c-b5a693575904","year":2023},"citing_paper":{"arxiv_id":"2605.07453","last_updated":"2026-05-08T09:00:04Z","snapshot_observed_at":"2026-08-11T15:25:22.585680Z","submitted_at":"2026-05-08T09:00:04Z","title":"Data Contamination in Neural Hieroglyphic Translation: A Reproducibility Study","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-05-11T01:55:16.674529Z"},"links":{"cited_paper":"/paper/2310.17623","citing_paper":"/paper/2605.07453"},"observation_digest":"sha256:baca9e892924e6af7d2e26d6528f8429176b8bd48af670ad9e251a722f672e8d","observation_id":"d31c1837-ca49-4caf-a2a3-d9d96da1dc5e","resolution":{"observed_at":"2026-05-11T04:10:55.459234Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17623","last_updated":"2023-11-24T01:45:16Z","snapshot_observed_at":"2026-07-06T16:39:08.850847Z","submitted_at":"2023-10-26T17:43:13Z","title":"Proving Test Set Contamination in Black Box Language Models","version":2},"cited_work":{"arxiv_id":"2310.17623","doi":"10.48550/arxiv.2310.17623","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17623","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatterji, Faisal Ladhak, and Tatsunori Hashimoto","venue":"arXiv (Cornell University)","work_id":"cf915c30-398f-4fa5-ad6c-b5a693575904","year":2023},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-11T17:02:03.648073Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2310.17623","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:0cf2be1dd2fba5b823840ebce7f71b749f087253fb910431c37fb1c2266fe5f3","observation_id":"824532d7-400d-40ed-8919-f33015637cc8","resolution":{"observed_at":"2026-05-14T20:32:56.905072Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17623","last_updated":"2023-11-24T01:45:16Z","snapshot_observed_at":"2026-07-06T16:39:08.850847Z","submitted_at":"2023-10-26T17:43:13Z","title":"Proving Test Set Contamination in Black Box Language Models","version":2},"cited_work":{"arxiv_id":"2310.17623","doi":"10.48550/arxiv.2310.17623","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17623","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatterji, Faisal Ladhak, and Tatsunori Hashimoto","venue":"arXiv (Cornell University)","work_id":"cf915c30-398f-4fa5-ad6c-b5a693575904","year":2023},"citing_paper":{"arxiv_id":"2605.17829","last_updated":"2026-05-18T04:03:18Z","snapshot_observed_at":"2026-08-11T04:15:46.387174Z","submitted_at":"2026-05-18T04:03:18Z","title":"Interactive Evaluation Requires a Design Science","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-20T10:55:08.135630Z"},"links":{"cited_paper":"/paper/2310.17623","citing_paper":"/paper/2605.17829"},"observation_digest":"sha256:7da65e2ec1e88825312a65d1841cc154cd8a70d91e94b34eb30f3cd5f1eb6eda","observation_id":"94f875a4-f10d-42f6-9998-136ddd67eb62","resolution":{"observed_at":"2026-05-20T10:58:14.121307Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17623","last_updated":"2023-11-24T01:45:16Z","snapshot_observed_at":"2026-07-06T16:39:08.850847Z","submitted_at":"2023-10-26T17:43:13Z","title":"Proving Test Set Contamination in Black Box Language Models","version":2},"cited_work":{"arxiv_id":"2310.17623","doi":"10.48550/arxiv.2310.17623","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17623","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatterji, Faisal Ladhak, and Tatsunori Hashimoto","venue":"arXiv (Cornell University)","work_id":"cf915c30-398f-4fa5-ad6c-b5a693575904","year":2023},"citing_paper":{"arxiv_id":"2605.26133","last_updated":"2026-05-21T10:32:33Z","snapshot_observed_at":"2026-08-07T10:44:40.983544Z","submitted_at":"2026-05-21T10:32:33Z","title":"Pretraining Data Exposure in Large Language Models: A Survey of Membership Inference, Data Contamination, and Security Implications","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-06-30T17:20:16.735285Z"},"links":{"cited_paper":"/paper/2310.17623","citing_paper":"/paper/2605.26133"},"observation_digest":"sha256:99c5e2b890e1fa970dbf21822156bc3eb16ff353f27ae3f4567f17b1eabcd0f4","observation_id":"27fcf841-9a4e-44ef-8d5b-7eda51f7cc89","resolution":{"observed_at":"2026-06-30T17:24:56.592669Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17623","last_updated":"2023-11-24T01:45:16Z","snapshot_observed_at":"2026-07-06T16:39:08.850847Z","submitted_at":"2023-10-26T17:43:13Z","title":"Proving Test Set Contamination in Black Box Language Models","version":2},"cited_work":{"arxiv_id":"2310.17623","doi":"10.48550/arxiv.2310.17623","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17623","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatterji, Faisal Ladhak, and Tatsunori Hashimoto","venue":"arXiv (Cornell University)","work_id":"cf915c30-398f-4fa5-ad6c-b5a693575904","year":2023},"citing_paper":{"arxiv_id":"2606.07996","last_updated":"2026-06-06T06:27:54Z","snapshot_observed_at":"2026-07-31T22:22:12.944552Z","submitted_at":"2026-06-06T06:27:54Z","title":"MC-PDD: Masked Corpus-Level Pretraining Data Detection for Black-Box Large Language Models","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-27T20:02:50.169589Z"},"links":{"cited_paper":"/paper/2310.17623","citing_paper":"/paper/2606.07996"},"observation_digest":"sha256:63e211d30e937cdaf47811c80d8e09beac33d4e070c3a84b1d0d355c3e9274c9","observation_id":"114e7f20-0f73-4110-a023-6717d66ddf4c","resolution":{"observed_at":"2026-07-02T20:57:23.345069Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17623","last_updated":"2023-11-24T01:45:16Z","snapshot_observed_at":"2026-07-06T16:39:08.850847Z","submitted_at":"2023-10-26T17:43:13Z","title":"Proving Test Set Contamination in Black Box Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.17623","snapshot_observed_at":"2026-08-08T04:31:04.382539Z","title":"Daniel Paleka, Shashwat Goel, Jonas Geiping, and Florian Tramèr","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-12T00:51:55.200790Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.382539Z"},"links":{"cited_paper":"/paper/2310.17623","citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:b88bf24a92bf55752407bb649164be87a34989cee1cc1d5003daa4dd014ce2e0","observation_id":"f10255f9-bef7-4875-8408-e3c823662b4e","resolution":{"observed_at":"2026-08-08T04:31:04.382539Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2310.17623/citation-record","integrity":"/paper/2310.17623/integrity","json":"/paper/2310.17623/citation-record.json","paper":"/paper/2310.17623"},"outbound":[],"paper":{"arxiv_id":"2310.17623","last_updated":"2023-11-24T01:45:16Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T16:39:08.850847Z","submitted_at":"2023-10-26T17:43:13Z","title":"Proving Test Set Contamination in Black Box Language Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 12 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 18 inbound Pith citation observations for arXiv:2310.17623."}