{"as_of":"2026-08-09T15:47:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:3c469604113e1a6d6c66ae3dfb71c2caebb58ee37556ac44e5c178b5f977c0e0","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":20,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":20,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":20,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":20,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T14:51:15.453440Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":8,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-08T14:51:15.453440Z","title":"A., Garc ´ıa-Ferrero, I., Etxaniz, J., de Lacalle, O","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.06655","last_updated":"2025-05-12T14:34:05Z","snapshot_observed_at":"2026-08-09T03:53:29.137552Z","submitted_at":"2025-02-10T16:45:18Z","title":"Unbiased Evaluation of Large Language Models from a Causal Perspective","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-08T14:51:15.453440Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2502.06655"},"observation_digest":"sha256:932ebe10619c5a3c7422716365d8b80c3fa734dc3d909a4ea6d4d29b916abef0","observation_id":"d38930be-bc42-42ce-8362-ea6b0e0573df","resolution":{"observed_at":"2026-08-08T14:51:15.453440Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-07T11:04:07.003448Z","title":"Nlp evaluation in trouble: On the need to measure llm data contamination for each benchmark,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.03637","last_updated":"2025-07-07T09:53:22Z","snapshot_observed_at":"2026-08-08T03:23:19.289933Z","submitted_at":"2025-06-04T07:30:16Z","title":"RewardAnything: Generalizable Principle-Following Reward Models","version":2},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-07T11:04:07.003448Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2506.03637"},"observation_digest":"sha256:12026719930e786a50b63619732de9cc64ed8748e38e4f9ba0a30b1502c5cb50","observation_id":"9171a74f-991e-4ff4-b18d-8ed9ca0db66c","resolution":{"observed_at":"2026-08-07T11:04:07.003448Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-07T10:54:26.345390Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.04142","last_updated":"2025-06-04T16:33:44Z","snapshot_observed_at":"2026-08-07T21:43:19.257599Z","submitted_at":"2025-06-04T16:33:44Z","title":"Establishing Trustworthy LLM Evaluation via Shortcut Neuron Analysis","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-07T10:54:26.345390Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2506.04142"},"observation_digest":"sha256:ed7dfee2c33d4a81836ef268dbe02faa3dbc224c849cfd02f257b40a3f3d2faa","observation_id":"6ff839f5-66d4-4b38-947c-22946783829d","resolution":{"observed_at":"2026-08-07T10:54:26.345390Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":"2310.18018","doi":"10.48550/arxiv.2310.18018","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nlp evaluation in trouble: On the need to measure llm data contamination for each benchmark","venue":"arXiv (Cornell University)","work_id":"feaf47be-d93d-4603-a19e-9dfb3378c90c","year":2023},"citing_paper":{"arxiv_id":"2507.22359","last_updated":"2026-04-14T11:47:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-30T03:50:46Z","title":"League of LLMs: A Benchmark-Free Paradigm for Mutual Evaluation of Large Language Models","version":4},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-05-19T03:17:06.457421Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2507.22359"},"observation_digest":"sha256:346b60ba786f77f1bd22dfca4afb35290c7067e96e44a331f569ddce377a34a0","observation_id":"67128382-2d9a-44be-83dd-fd934e1ae9bd","resolution":{"observed_at":"2026-05-19T03:22:01.356975Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-05T19:28:54.967333Z","title":"Nlp evaluation in trouble: On the need to measure llm data contamination for each benchmark,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.12464","last_updated":"2025-08-17T18:27:54Z","snapshot_observed_at":"2026-08-06T19:31:58.152011Z","submitted_at":"2025-08-17T18:27:54Z","title":"On the Fitness Landscape in the $NK$ Model","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-05T19:28:54.967333Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2508.12464"},"observation_digest":"sha256:b0434bf96687ad91fff35361bde6a5513d4a1a1150280626a30d682dc82b0c33","observation_id":"0f330cad-9e76-467d-95ef-727506a63474","resolution":{"observed_at":"2026-08-05T19:28:54.967333Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":"2310.18018","doi":"10.48550/arxiv.2310.18018","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nlp evaluation in trouble: On the need to measure llm data contamination for each benchmark","venue":"arXiv (Cornell University)","work_id":"feaf47be-d93d-4603-a19e-9dfb3378c90c","year":2023},"citing_paper":{"arxiv_id":"2509.23108","last_updated":"2026-05-19T05:14:38Z","snapshot_observed_at":"2026-07-06T22:30:55.313733Z","submitted_at":"2025-09-27T04:36:12Z","title":"Artificial Phantasia: Emergent Mental Imagery in Large Language Models","version":2},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-05-21T21:41:39.111769Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2509.23108"},"observation_digest":"sha256:b6763da29fc3a1775f110800f625ce5190bceda7c74b68c78224ab7b7bde5786","observation_id":"56c0d6cd-85ae-444a-96c4-7198018eabf1","resolution":{"observed_at":"2026-05-21T21:44:22.869054Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-04T11:19:49.706763Z","title":"NLP evaluation in trouble: On the need to measure LLM data contamination for each benchmark, December 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2510.05709","last_updated":"2026-06-04T12:15:57Z","snapshot_observed_at":"2026-08-07T08:05:26.965055Z","submitted_at":"2025-10-07T09:22:22Z","title":"Correcting Prompt Dependence in LLM Benchmarks: A Bayesian Hierarchical Model with Embedding-Space Clustering","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-04T11:19:49.706763Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2510.05709"},"observation_digest":"sha256:8a1dac7d18893d344997fb6ebb8d1ba560f8ae8e68da45cf2c2348ceed9bd9e9","observation_id":"e5b6228e-fa83-4191-b596-db823287d718","resolution":{"observed_at":"2026-08-04T11:19:49.706763Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":"2310.18018","doi":"10.48550/arxiv.2310.18018","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nlp evaluation in trouble: On the need to measure llm data contamination for each benchmark","venue":"arXiv (Cornell University)","work_id":"feaf47be-d93d-4603-a19e-9dfb3378c90c","year":2023},"citing_paper":{"arxiv_id":"2511.02627","last_updated":"2026-06-11T15:36:40Z","snapshot_observed_at":"2026-08-08T22:40:29.987506Z","submitted_at":"2025-11-04T14:57:11Z","title":"DecompSR: A dataset for decomposed analyses of compositional multihop spatial reasoning","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-18T01:18:44.523602Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2511.02627"},"observation_digest":"sha256:1593eb5e557750c93c3b87fb8e9e53da473038d943b44fe02e30c0e9242788ee","observation_id":"293cbe29-27d6-4e41-91e6-87ae083e33c7","resolution":{"observed_at":"2026-05-18T01:20:34.337660Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-04T00:11:51.863397Z","title":"Nlp evaluation in trouble: On the need to measure llm data contamination for each benchmark","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2511.02627","last_updated":"2026-06-11T15:36:40Z","snapshot_observed_at":"2026-08-08T22:40:29.987506Z","submitted_at":"2025-11-04T14:57:11Z","title":"DecompSR: A dataset for decomposed analyses of compositional multihop spatial reasoning","version":4},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-04T00:11:51.863397Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2511.02627"},"observation_digest":"sha256:b0c75234ddb83475613c9365cc589c839f6585a7187bac752b7f902df27a01c2","observation_id":"5fe6ffde-e7d3-4b18-9162-da15dd263a19","resolution":{"observed_at":"2026-08-04T00:11:51.863397Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-03T06:46:26.337035Z","title":"Nlp evaluation in trouble: On the need to measure llm data contamination for each benchmark.arXiv preprint arXiv:2310.18018, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2601.22025","last_updated":"2026-06-09T23:57:32Z","snapshot_observed_at":"2026-08-07T21:40:38.906811Z","submitted_at":"2026-01-29T17:32:34Z","title":"When Generic Prompt Improvements Hurt: Evaluation-Driven Iteration for LLM Applications","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-03T06:46:26.337035Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2601.22025"},"observation_digest":"sha256:bae1c5780c9c1cb340fb4a5705be25b1b5ee69569fff0a578af89167a54442aa","observation_id":"d6048266-81af-4ed5-9c79-d42da57ac594","resolution":{"observed_at":"2026-08-03T06:46:26.337035Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":"2310.18018","doi":"10.48550/arxiv.2310.18018","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nlp evaluation in trouble: On the need to measure llm data contamination for each benchmark","venue":"arXiv (Cornell University)","work_id":"feaf47be-d93d-4603-a19e-9dfb3378c90c","year":2023},"citing_paper":{"arxiv_id":"2604.18955","last_updated":"2026-04-21T01:05:52Z","snapshot_observed_at":"2026-07-06T23:05:44.499005Z","submitted_at":"2026-04-21T01:05:52Z","title":"Assessing Capabilities of Large Language Models in Social Media Analytics: A Multi-task Quest","version":1},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-05-10T03:20:12.777878Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2604.18955"},"observation_digest":"sha256:7b510fee92ec5525722b9e7bd1f599ef9b529d35badaf43b3ffb037ca5fb61c9","observation_id":"14e051e7-4ac8-4f34-8c88-9452415addd0","resolution":{"observed_at":"2026-05-11T12:41:01.820540Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":"2310.18018","doi":"10.48550/arxiv.2310.18018","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nlp evaluation in trouble: On the need to measure llm data contamination for each benchmark","venue":"arXiv (Cornell University)","work_id":"feaf47be-d93d-4603-a19e-9dfb3378c90c","year":2023},"citing_paper":{"arxiv_id":"2604.20273","last_updated":"2026-04-22T07:20:03Z","snapshot_observed_at":"2026-07-06T23:06:44.624357Z","submitted_at":"2026-04-22T07:20:03Z","title":"ActuBench: A Multi-Agent LLM Pipeline for Generation and Evaluation of Actuarial Reasoning Tasks","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-10T00:35:24.397273Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2604.20273"},"observation_digest":"sha256:63eacb39131d52cef91269648c854d0aac4390f2aadcea20b509d5b86a8a0005","observation_id":"0c62ac00-20a6-47ee-9c7c-338a0ba8f1a1","resolution":{"observed_at":"2026-05-11T13:46:05.187194Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":"2310.18018","doi":"10.48550/arxiv.2310.18018","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nlp evaluation in trouble: On the need to measure llm data contamination for each benchmark","venue":"arXiv (Cornell University)","work_id":"feaf47be-d93d-4603-a19e-9dfb3378c90c","year":2023},"citing_paper":{"arxiv_id":"2605.24213","last_updated":"2026-05-22T20:54:30Z","snapshot_observed_at":"2026-08-02T06:58:01.300783Z","submitted_at":"2026-05-22T20:54:30Z","title":"Towards Evaluation Engineering: An Empirical Study of ML Evaluation Harnesses in the Wild","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-06-30T14:41:07.354007Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2605.24213"},"observation_digest":"sha256:c78695e07307858723793d799f5ff17f28cb8e5c1a744947b1b17ea488086c65","observation_id":"69286f6b-21dd-4371-9c4e-7f613ba6fab7","resolution":{"observed_at":"2026-06-30T14:44:45.135213Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":"2310.18018","doi":"10.48550/arxiv.2310.18018","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nlp evaluation in trouble: On the need to measure llm data contamination for each benchmark","venue":"arXiv (Cornell University)","work_id":"feaf47be-d93d-4603-a19e-9dfb3378c90c","year":2023},"citing_paper":{"arxiv_id":"2605.26133","last_updated":"2026-05-21T10:32:33Z","snapshot_observed_at":"2026-08-07T10:44:40.983544Z","submitted_at":"2026-05-21T10:32:33Z","title":"Pretraining Data Exposure in Large Language Models: A Survey of Membership Inference, Data Contamination, and Security Implications","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-06-30T17:20:16.735285Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2605.26133"},"observation_digest":"sha256:473571004fcbcef2533259197dd2f0bed0989aeb2381fe4db4b7a5366aa2c304","observation_id":"66eb410d-85f7-4e23-b259-15edd22a1667","resolution":{"observed_at":"2026-06-30T17:24:56.613744Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":"2310.18018","doi":"10.48550/arxiv.2310.18018","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nlp evaluation in trouble: On the need to measure llm data contamination for each benchmark","venue":"arXiv (Cornell University)","work_id":"feaf47be-d93d-4603-a19e-9dfb3378c90c","year":2023},"citing_paper":{"arxiv_id":"2605.26781","last_updated":"2026-05-26T09:50:35Z","snapshot_observed_at":"2026-07-06T23:36:34.652096Z","submitted_at":"2026-05-26T09:50:35Z","title":"LiveK12Bench: Have Large Multimodal Models Truly Conquered High School-level Examinations?","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-29T17:33:03.397468Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2605.26781"},"observation_digest":"sha256:5c68e213cacfd20b69074047d93db6210ad2556f418f380103cb52e547bbfed2","observation_id":"8bbe3091-2d39-4119-8607-e5f378ccfd67","resolution":{"observed_at":"2026-06-29T17:33:45.019663Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":"2310.18018","doi":"10.48550/arxiv.2310.18018","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nlp evaluation in trouble: On the need to measure llm data contamination for each benchmark","venue":"arXiv (Cornell University)","work_id":"feaf47be-d93d-4603-a19e-9dfb3378c90c","year":2023},"citing_paper":{"arxiv_id":"2606.01189","last_updated":"2026-05-31T12:11:47Z","snapshot_observed_at":"2026-08-06T00:50:49.461826Z","submitted_at":"2026-05-31T12:11:47Z","title":"The Case for Model Science: Verify, Explore, Steer, Refine","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-06-28T17:24:32.311565Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2606.01189"},"observation_digest":"sha256:28bd54688890ad7c8462f160e6706fc0c8f9b17dd9276760189f95fdb720d975","observation_id":"8ba959b7-1a87-4e52-8441-eaf435884bdf","resolution":{"observed_at":"2026-07-01T21:16:13.563798Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":"2310.18018","doi":"10.48550/arxiv.2310.18018","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nlp evaluation in trouble: On the need to measure llm data contamination for each benchmark","venue":"arXiv (Cornell University)","work_id":"feaf47be-d93d-4603-a19e-9dfb3378c90c","year":2023},"citing_paper":{"arxiv_id":"2606.17454","last_updated":"2026-06-17T04:51:06Z","snapshot_observed_at":"2026-08-02T05:33:09.695431Z","submitted_at":"2026-06-16T03:17:03Z","title":"Dissecting model behavior through agent trajectories","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-06-27T01:27:39.812496Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2606.17454"},"observation_digest":"sha256:59816fe874561622f43970a3abbe7f153d1125b6343f82a11a74812611746eda","observation_id":"66ad5f14-c0ad-47f2-80ba-698b5e72f080","resolution":{"observed_at":"2026-07-03T20:18:56.917233Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":"2310.18018","doi":"10.48550/arxiv.2310.18018","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nlp evaluation in trouble: On the need to measure llm data contamination for each benchmark","venue":"arXiv (Cornell University)","work_id":"feaf47be-d93d-4603-a19e-9dfb3378c90c","year":2023},"citing_paper":{"arxiv_id":"2607.00276","last_updated":"2026-06-30T23:52:15Z","snapshot_observed_at":"2026-08-09T09:31:38.270841Z","submitted_at":"2026-06-30T23:52:15Z","title":"Testing Frontier Large Language Models' Physics Literacy in Parallel Physical Worlds","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-07-02T19:18:43.558804Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2607.00276"},"observation_digest":"sha256:9a6f3c2b8941e8004066bdfbc683349344e9847d46654ee257fb83c992b0c425","observation_id":"17835370-b4a7-4b61-bee8-ff4db2e32631","resolution":{"observed_at":"2026-07-02T19:27:18.679688Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":"2310.18018","doi":"10.48550/arxiv.2310.18018","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nlp evaluation in trouble: On the need to measure llm data contamination for each benchmark","venue":"arXiv (Cornell University)","work_id":"feaf47be-d93d-4603-a19e-9dfb3378c90c","year":2023},"citing_paper":{"arxiv_id":"2607.01829","last_updated":"2026-07-02T07:49:55Z","snapshot_observed_at":"2026-07-07T00:07:19.026006Z","submitted_at":"2026-07-02T07:49:55Z","title":"Pre-Flight: A Benchmark for Evaluating Large Language Models on Aviation Operational Knowledge","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-03T13:36:31.189451Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2607.01829"},"observation_digest":"sha256:88ea71315ba23d3a1100f1e07ab7e687b20ce69e6375e4ea7c7ecf16dde4816f","observation_id":"4482a391-6e89-4f76-9714-9d72df27b199","resolution":{"observed_at":"2026-07-03T13:38:18.445574Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-02T13:26:26.237664Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark, October 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.19355","last_updated":"2026-05-22T12:13:16Z","snapshot_observed_at":"2026-08-07T21:44:05.596114Z","submitted_at":"2026-05-22T12:13:16Z","title":"Information Discernment in Large Language Models","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-02T13:26:26.237664Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2607.19355"},"observation_digest":"sha256:3e08ca9f04c987faea3bbf91b737d3373eea38966b42e276cad5b95462278eac","observation_id":"f07d090d-375a-4152-ad26-b306a029ab47","resolution":{"observed_at":"2026-08-02T13:26:26.237664Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2310.18018/citation-record","integrity":"/paper/2310.18018/integrity","json":"/paper/2310.18018/citation-record.json","paper":"/paper/2310.18018"},"outbound":[],"paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 20 inbound Pith citation observations for arXiv:2310.18018."}