{"as_of":"2026-08-09T03:24:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:4b38de747505265c70bd549759045f337fd4b38253b5f8b060fd44ad8b37bb6d","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":42,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":42,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":42,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":42,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T18:19:46.939354Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":7,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2406.04244","last_updated":"2024-06-06T16:41:39Z","snapshot_observed_at":"2026-07-30T15:43:06.151242Z","submitted_at":"2024-06-06T16:41:39Z","title":"Benchmark Data Contamination of Large Language Models: A Survey","version":1},"reference_index":170,"source":"pdf_text","source_observed_at":"2026-05-22T23:10:40.420241Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2406.04244"},"observation_digest":"sha256:ab489aa501aafffd31cf1959f3e573bc17f234188e4e9e59fa962b09b357f9e3","observation_id":"26dde097-e60d-47b9-ba7b-9ca5f4315064","resolution":{"observed_at":"2026-05-22T23:10:41.134008Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2406.11794","last_updated":"2025-04-21T17:48:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-17T17:42:57Z","title":"DataComp-LM: In search of the next generation of training sets for language models","version":4},"reference_index":208,"source":"pdf_text","source_observed_at":"2026-05-17T22:58:16.523267Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2406.11794"},"observation_digest":"sha256:979cf082e694acedba3e8d11c029a3798229c71069fb85bd82a9004dd6767283","observation_id":"79ffe552-61e5-4673-bf9b-bc4fde5114c9","resolution":{"observed_at":"2026-05-17T22:58:17.334977Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2406.12793","last_updated":"2024-07-30T03:58:11Z","snapshot_observed_at":"2026-08-07T13:56:34.167869Z","submitted_at":"2024-06-18T16:58:21Z","title":"ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-11T08:08:09.444352Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2406.12793"},"observation_digest":"sha256:578a505cc8f9088aaf1edc718790865a9640693b8772a3b67db4aa0df4695235","observation_id":"d04fd98e-06bd-41e4-af68-e6b519a0ec47","resolution":{"observed_at":"2026-05-11T08:08:09.653347Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-08T15:06:55.384462Z","title":"Gonzalez, and Ion Stoica","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-08T23:28:10.897343Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":136,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.384462Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:6ab2547b8a0562ce8c9b67ff2969d8bcf1dfbb8b4ab32f7c1003d271bf615324","observation_id":"721c1b03-93f3-4b00-a4dc-c92cdd4ae55f","resolution":{"observed_at":"2026-08-08T15:06:55.384462Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-08T14:51:15.489455Z","title":"E., and Stoica, I","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.06655","last_updated":"2025-05-12T14:34:05Z","snapshot_observed_at":"2026-08-08T14:44:11.158498Z","submitted_at":"2025-02-10T16:45:18Z","title":"Unbiased Evaluation of Large Language Models from a Causal Perspective","version":2},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-08T14:51:15.489455Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2502.06655"},"observation_digest":"sha256:63a8a43706e92ef9c2cf5d613240317fb37f18b3ef9512911589a470243931bb","observation_id":"246859ee-fbaf-4caa-ba72-24d790e7d46a","resolution":{"observed_at":"2026-08-08T14:51:15.489455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-07T14:44:35.529969Z","title":"Gonzalez, and Ion Stoica","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.18102","last_updated":"2026-05-30T13:29:55Z","snapshot_observed_at":"2026-08-07T14:33:15.240605Z","submitted_at":"2025-05-23T16:57:34Z","title":"CapBencher: Give Your LLM Benchmark a Built-in Alarm for Test-Set Overfitting","version":7},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T14:44:35.529969Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2505.18102"},"observation_digest":"sha256:bdec4b7fbc1408b5c475a73ba02a469f151778833dc7afda5a472c4aa5b9a44c","observation_id":"775566c9-a666-4c3c-a592-902782fcb963","resolution":{"observed_at":"2026-08-07T14:44:35.529969Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-07T12:29:18.525978Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.24324","last_updated":"2025-05-30T08:06:30Z","snapshot_observed_at":"2026-08-08T23:46:24.421775Z","submitted_at":"2025-05-30T08:06:30Z","title":"SwiftEval: Developing a Language-Specific Benchmark for LLM-generated Code Evaluation","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T12:29:18.525978Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2505.24324"},"observation_digest":"sha256:45845cdbd096c0f51c412856f84799d97b9b31a386eb4e52e90574cabe8ae93a","observation_id":"fb1ad3e0-5e03-4ab1-aeff-d7c4f5f9f56d","resolution":{"observed_at":"2026-08-07T12:29:18.525978Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-07T12:08:17.369531Z","title":"Rethinking benchmark and contamination for language models with rephrased samples.arXiv preprint arXiv:2311.04850, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.00482","last_updated":"2025-05-31T09:24:32Z","snapshot_observed_at":"2026-08-07T12:01:57.559855Z","submitted_at":"2025-05-31T09:24:32Z","title":"BenchHub: A Unified Benchmark Suite for Holistic and Customizable LLM Evaluation","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-07T12:08:17.369531Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2506.00482"},"observation_digest":"sha256:a100d5d481d4c7e71944c2b0048e215e196234a6f0aadace65e23db2e46ca261","observation_id":"0a0588e0-0c37-4bd2-95fb-f1c7e168d36e","resolution":{"observed_at":"2026-08-07T12:08:17.369531Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-06T18:11:20.143746Z","title":", author Chiang, W.L","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09075","last_updated":"2025-07-11T23:35:54Z","snapshot_observed_at":"2026-08-08T12:25:01.827357Z","submitted_at":"2025-07-11T23:35:54Z","title":"OpenCodeReasoning-II: A Simple Test Time Scaling Approach via Self-Critique","version":1},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-06T18:11:20.143746Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2507.09075"},"observation_digest":"sha256:e19c063c8ebfe17aa637b29ae79f1fdfb8f24993f26c62f9c00cdcfaf7ee2b03","observation_id":"b690f495-6cb2-4638-a9e5-a77112b77557","resolution":{"observed_at":"2026-08-06T18:11:20.143746Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2507.22359","last_updated":"2026-04-14T11:47:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-30T03:50:46Z","title":"League of LLMs: A Benchmark-Free Paradigm for Mutual Evaluation of Large Language Models","version":4},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-05-19T03:17:06.457421Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2507.22359"},"observation_digest":"sha256:3922873e5a9f7e11b52f01ad6397aa02e1175299650a571e3682cde53af2e4ee","observation_id":"1bb22ce2-c68b-422d-948d-2f1f77c32e39","resolution":{"observed_at":"2026-05-19T03:22:01.341453Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-06T05:57:29.511258Z","title":"Gonzalez, and Ion Stoica","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.01059","last_updated":"2025-08-01T20:25:57Z","snapshot_observed_at":"2026-08-06T12:22:28.743019Z","submitted_at":"2025-08-01T20:25:57Z","title":"Llama-3.1-FoundationAI-SecurityLLM-8B-Instruct Technical Report","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-06T05:57:29.511258Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2508.01059"},"observation_digest":"sha256:8aefd4bc3ac85309d3b9e03e35a2fee0a93041349f0399633af474576437cb5e","observation_id":"8fca8ffb-2cd4-4adf-b168-b5219a677eb7","resolution":{"observed_at":"2026-08-06T05:57:29.511258Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2509.20909","last_updated":"2026-05-09T19:01:47Z","snapshot_observed_at":"2026-08-02T22:05:17.866277Z","submitted_at":"2025-09-25T08:55:18Z","title":"LogitTrace: Detecting Benchmark Contamination via Layerwise Logit Trajectories","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-05-18T14:32:57.426996Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2509.20909"},"observation_digest":"sha256:c0cf0771b022366f676bd0e83a4d855b98c8dc0561f2e3f1b403f421eeabe41a","observation_id":"54c99d7c-0932-4665-88df-4cafaffd7561","resolution":{"observed_at":"2026-05-18T14:36:28.943124Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-04T11:19:51.012398Z","title":"Rethinking benchmark and contam- ination for language models with rephrased samples, November 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2510.05709","last_updated":"2026-06-04T12:15:57Z","snapshot_observed_at":"2026-08-07T08:05:26.965055Z","submitted_at":"2025-10-07T09:22:22Z","title":"Correcting Prompt Dependence in LLM Benchmarks: A Bayesian Hierarchical Model with Embedding-Space Clustering","version":2},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-04T11:19:51.012398Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2510.05709"},"observation_digest":"sha256:870270dad037d2e5a4a31fd19904a046152e06490965957e212489cda47cacde","observation_id":"1cf0bda5-0b6e-4ddb-b0ff-0793c0142615","resolution":{"observed_at":"2026-08-04T11:19:51.012398Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2604.05150","last_updated":"2026-04-06T20:25:20Z","snapshot_observed_at":"2026-07-06T22:53:55.926521Z","submitted_at":"2026-04-06T20:25:20Z","title":"Compiled AI: Deterministic Code Generation for LLM-Based Workflow Automation","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-10T18:57:34.386632Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2604.05150"},"observation_digest":"sha256:a54c665ab02dc18cc5dda022221abc0653f9cc8d0aa86a32ea350c4fbaa85faa","observation_id":"d25e5c67-f333-4fda-b471-faf6e2b0bdff","resolution":{"observed_at":"2026-05-10T23:40:51.520975Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2604.09251","last_updated":"2026-04-23T09:41:06Z","snapshot_observed_at":"2026-07-06T22:58:08.579543Z","submitted_at":"2026-04-10T12:07:22Z","title":"DRBENCHER: Can Your Agent Identify the Entity, Retrieve Its Properties and Do the Math?","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T16:46:24.780270Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2604.09251"},"observation_digest":"sha256:64d65ea9be08ff9903e6dc991d7e18598ece4db76e4acb9a971da0992feaa75d","observation_id":"548a2693-0ac5-442b-aac0-89d3bec954ea","resolution":{"observed_at":"2026-05-11T08:11:04.832167Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2604.17966","last_updated":"2026-04-20T08:46:49Z","snapshot_observed_at":"2026-08-01T00:56:32.861823Z","submitted_at":"2026-04-20T08:46:49Z","title":"TPS-CalcBench: A Benchmark and Diagnostic Evaluation Framework for LLM Analytical Calculation Competence in Hypersonic Thermal Protection System Engineering","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-05-10T05:25:59.343884Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2604.17966"},"observation_digest":"sha256:153a5b0958ba37b0c783baf902268cc5a74d6b71a124c4a3793f5bc0fb41b265","observation_id":"47e2479c-fb1f-496d-827e-eaaf883b2921","resolution":{"observed_at":"2026-05-10T09:23:37.514127Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2604.24712","last_updated":"2026-04-27T17:21:09Z","snapshot_observed_at":"2026-07-06T23:10:42.926677Z","submitted_at":"2026-04-27T17:21:09Z","title":"When Prompt Under-Specification Improves Code Correctness: An Exploratory Study of Prompt Wording and Structure Effects on LLM-Based Code Generation","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-08T03:00:26.137401Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2604.24712"},"observation_digest":"sha256:be9bca9e39f4e363b80fb1584f53a356766dfa48a4137aec4966040da534102c","observation_id":"22b7cd4d-8090-460d-9758-5f01a02de39b","resolution":{"observed_at":"2026-05-11T22:17:06.819446Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.02442","last_updated":"2026-05-04T10:42:26Z","snapshot_observed_at":"2026-07-30T10:52:38.632184Z","submitted_at":"2026-05-04T10:42:26Z","title":"Measuring AI Reasoning: A Guide for Researchers","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-05-08T18:53:18.586923Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.02442"},"observation_digest":"sha256:47ed6dd7b0c33c20d891095680e819e9848673084101402dfa4140b8af7c7079","observation_id":"6fd5b37e-9661-4f13-97f0-0af6e7d0b045","resolution":{"observed_at":"2026-05-09T06:05:36.959600Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.04312","last_updated":"2026-05-05T21:24:58Z","snapshot_observed_at":"2026-07-06T23:17:04.200299Z","submitted_at":"2026-05-05T21:24:58Z","title":"Agent Island: A Saturation- and Contamination-Resistant Benchmark from Multiagent Games","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-05-08T17:06:32.814188Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.04312"},"observation_digest":"sha256:526b974876176077bf3c2dbcee551e048eafccf545641c9c4602e1ec9588cee7","observation_id":"1b15f2a6-e35a-4bde-a1ba-9bf71adfead8","resolution":{"observed_at":"2026-05-11T17:46:17.276350Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.06327","last_updated":"2026-05-07T14:23:31Z","snapshot_observed_at":"2026-07-06T23:18:51.157048Z","submitted_at":"2026-05-07T14:23:31Z","title":"Measuring Evaluation-Context Divergence in Open-Weight LLMs: A Paired-Prompt Protocol with Pilot Evidence of Alignment-Pipeline-Specific Heterogeneity","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-05-08T10:23:02.697982Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.06327"},"observation_digest":"sha256:f80d4502af0ed1922d60be23a74bb508ad11ea81b42b99a9c28f1cb1e10da566","observation_id":"9d53cd78-ddd4-4c96-9571-3b76c9fa1010","resolution":{"observed_at":"2026-05-08T22:04:18.090953Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.07053","last_updated":"2026-05-26T07:47:31Z","snapshot_observed_at":"2026-08-02T22:34:50.237073Z","submitted_at":"2026-05-08T00:02:39Z","title":"GSM-SEM: Benchmark and Framework for Generating Semantically Variant Augmentations","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-05-11T00:57:55.616036Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.07053"},"observation_digest":"sha256:248ceda930dbf88eeaa3ea2db2b697e0eb438fda07cc3b5d7408ba93f985854d","observation_id":"1c5fe167-2a6f-4ffc-b264-cb3a266a33fe","resolution":{"observed_at":"2026-05-11T04:55:59.899239Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.07053","last_updated":"2026-05-26T07:47:31Z","snapshot_observed_at":"2026-08-02T22:34:50.237073Z","submitted_at":"2026-05-08T00:02:39Z","title":"GSM-SEM: Benchmark and Framework for Generating Semantically Variant Augmentations","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-06-30T23:42:00.965554Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.07053"},"observation_digest":"sha256:c4f16ff979c0e323c885bc4e9925dd167ca2cac6eaf62eae13fb28cb0b20a694","observation_id":"bcaeda91-673a-474a-8963-2564111db7eb","resolution":{"observed_at":"2026-06-30T23:45:07.976186Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.10448","last_updated":"2026-05-11T12:20:15Z","snapshot_observed_at":"2026-08-06T20:03:57.191605Z","submitted_at":"2026-05-11T12:20:15Z","title":"Can Agent Benchmarks Support Their Scores? Evidence-Supported Bounds for Interactive-Agent Evaluation","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-12T05:05:55.592359Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.10448"},"observation_digest":"sha256:dea67df1e2036ad1299fef36b0dc7d21c18e72ded07bf6838479f16adc70ed36","observation_id":"ec4241e4-e7c0-4b4c-8c06-4660896cd370","resolution":{"observed_at":"2026-05-12T05:41:23.900869Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.11501","last_updated":"2026-05-12T04:21:26Z","snapshot_observed_at":"2026-07-06T23:23:21.034839Z","submitted_at":"2026-05-12T04:21:26Z","title":"Decaf: Improving Neural Decompilation with Automatic Feedback and Search","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-13T02:03:19.836506Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.11501"},"observation_digest":"sha256:9c0e7bc025994d3f0ccdb7120d80d3d8109592817b7e972477d67629a713e81a","observation_id":"abfb013e-e714-4037-85ac-acc881089016","resolution":{"observed_at":"2026-05-13T02:07:08.093098Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-02T02:31:32.394072Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:892023241eece73ae2befe17d68c0e22f1b8dc806775fc34d51e1c257b018204","observation_id":"481aa828-3a0e-498f-9f0b-f56df90e7622","resolution":{"observed_at":"2026-05-14T20:32:56.862291Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.19999","last_updated":"2026-05-19T15:33:16Z","snapshot_observed_at":"2026-07-06T23:30:39.512029Z","submitted_at":"2026-05-19T15:33:16Z","title":"LLM Benchmark Datasets Should Be Contamination-Resistant","version":1},"reference_index":90,"source":"arxiv_source","source_observed_at":"2026-05-20T07:19:50.354875Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.19999"},"observation_digest":"sha256:783a0fbd49606894b58e1ffa04037f84d0a5f13d0a7294de2e54c6e7f398326a","observation_id":"a0f175ac-6597-48d7-9bf4-51fc3977f5ec","resolution":{"observed_at":"2026-05-20T07:23:07.058655Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.21543","last_updated":"2026-05-20T09:16:39Z","snapshot_observed_at":"2026-08-01T19:04:50.092578Z","submitted_at":"2026-05-20T09:16:39Z","title":"Provable Joint Decontamination for Benchmarking Multiple Large Language Models","version":1},"reference_index":172,"source":"arxiv_source","source_observed_at":"2026-05-22T00:40:54.038367Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.21543"},"observation_digest":"sha256:9bde49c50c36ab57117a2eb9b76b36fd0753b264aec5f6fed8147feadc02bcf3","observation_id":"ae61037a-4a0c-4149-9764-5047fc27a22e","resolution":{"observed_at":"2026-05-22T00:44:29.489044Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.21856","last_updated":"2026-05-21T01:06:19Z","snapshot_observed_at":"2026-08-02T10:22:40.169491Z","submitted_at":"2026-05-21T01:06:19Z","title":"The Illusion of Reasoning: Exposing Evasive Data Contamination in LLMs via Zero-CoT Truncation","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-22T08:05:42.212459Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.21856"},"observation_digest":"sha256:57e7c9704508a33ff6e95c6d0cb6c38ba642da07962de3628b04aee701c8d55b","observation_id":"e50e38d5-a330-4082-9fc1-b47897ecc449","resolution":{"observed_at":"2026-05-22T08:06:15.310194Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.23628","last_updated":"2026-05-22T13:40:00Z","snapshot_observed_at":"2026-07-06T23:33:49.013121Z","submitted_at":"2026-05-22T13:40:00Z","title":"How Hard is it to Rig a Benchmark? A Social Choice Analysis of Leaderboard Robustness","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-25T04:37:01.537128Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.23628"},"observation_digest":"sha256:2983bfbfd6e481532f9cfe0c85a90fbcc72694ca6da0b5936b6a07274cea6da5","observation_id":"81a8cf3c-06b8-49b8-bbd7-3da728bad548","resolution":{"observed_at":"2026-05-25T04:40:24.444660Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.24079","last_updated":"2026-05-22T17:30:20Z","snapshot_observed_at":"2026-08-01T09:58:43.814751Z","submitted_at":"2026-05-22T17:30:20Z","title":"TRACER: A Semantic-Aware Framework for Fine-Grained Contamination Detection in Code LLMs","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-06-30T15:32:05.976335Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.24079"},"observation_digest":"sha256:578c75f7e90742e62c1984e4d01465ccda61ae493057f5873a0e5bada45a9525","observation_id":"e11c3d40-017b-429b-8690-ff2370e45d98","resolution":{"observed_at":"2026-06-30T15:34:47.895008Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.24213","last_updated":"2026-05-22T20:54:30Z","snapshot_observed_at":"2026-08-02T06:58:01.300783Z","submitted_at":"2026-05-22T20:54:30Z","title":"Towards Evaluation Engineering: An Empirical Study of ML Evaluation Harnesses in the Wild","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-06-30T14:41:07.354007Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.24213"},"observation_digest":"sha256:b48d99cd887a14c776d61708d5a3745e0dd0478deec8f750cecc6cfbe224c7ec","observation_id":"04ea9e21-1c59-4575-a510-f812fd2b7aeb","resolution":{"observed_at":"2026-06-30T14:44:45.155954Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.24661","last_updated":"2026-07-01T08:31:40Z","snapshot_observed_at":"2026-08-03T01:04:22.766287Z","submitted_at":"2026-05-23T17:03:42Z","title":"Measuring Reasoning Quality in LLMs: A Multi-Dimensional Behavioral Framework","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-06-30T13:27:50.367497Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.24661"},"observation_digest":"sha256:5e98fa674613234badd76d2c86263a77ec3ce70e32b49ea3f5cfe9adc1f4f5db","observation_id":"6c2dc137-5f7d-4787-b49a-71cbffd49ef8","resolution":{"observed_at":"2026-06-30T13:34:40.588543Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.24661","last_updated":"2026-07-01T08:31:40Z","snapshot_observed_at":"2026-08-03T01:04:22.766287Z","submitted_at":"2026-05-23T17:03:42Z","title":"Measuring Reasoning Quality in LLMs: A Multi-Dimensional Behavioral Framework","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-07-01T07:35:51.017797Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.24661"},"observation_digest":"sha256:c31d886f56ef442155e865d001befbfd2bfcc6f11461631148d551419553aa6b","observation_id":"b5a659b1-05df-4a48-8f38-913a67b3cfe8","resolution":{"observed_at":"2026-07-01T08:05:31.914578Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.24661","last_updated":"2026-07-01T08:31:40Z","snapshot_observed_at":"2026-08-03T01:04:22.766287Z","submitted_at":"2026-05-23T17:03:42Z","title":"Measuring Reasoning Quality in LLMs: A Multi-Dimensional Behavioral Framework","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-07-02T23:22:08.567841Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.24661"},"observation_digest":"sha256:8af7dd22e232c5d14b010bbc90e3e188fc91d8f4c8c396ec059a892b9f87a40e","observation_id":"73c90e48-132c-4c28-9ee1-a4949e9206f0","resolution":{"observed_at":"2026-07-02T23:27:26.731890Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.26161","last_updated":"2026-05-24T14:59:12Z","snapshot_observed_at":"2026-08-02T04:25:25.218976Z","submitted_at":"2026-05-24T14:59:12Z","title":"TSFMAudit: Data Contamination Auditing in Forecasting Time Series Foundation Models","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-30T12:00:05.807004Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.26161"},"observation_digest":"sha256:c7f44d8f017103d63a9f93b57ccc1d71e20009ce77db53bf29e6520524579fe8","observation_id":"dab2454a-666d-4842-b07a-aed17edf3fd0","resolution":{"observed_at":"2026-06-30T12:04:38.780577Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2606.11909","last_updated":"2026-06-10T10:37:27Z","snapshot_observed_at":"2026-08-06T21:39:03.026837Z","submitted_at":"2026-06-10T10:37:27Z","title":"Embodied-BenchClaw: An Autonomous Multi-Agent System for Embodied Spatial Intelligence Benchmark Construction","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-27T09:48:49.786021Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2606.11909"},"observation_digest":"sha256:12249e030ffa78e7d7c25bf26fbff6c1b72667073722c02de07d207bcbf6ecb6","observation_id":"429ec9d5-66e8-4def-a403-8144313ea449","resolution":{"observed_at":"2026-07-03T10:48:02.880656Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2606.12385","last_updated":"2026-06-10T17:47:59Z","snapshot_observed_at":"2026-08-05T09:55:05.283277Z","submitted_at":"2026-06-10T17:47:59Z","title":"Which Models Are Our Models Built On? Auditing Invisible Dependencies in Modern LLMs","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-06-27T09:57:14.328157Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2606.12385"},"observation_digest":"sha256:518deb86f83b0d4b6f21cc05095a751a3745cfbb1bbe186479e1422455a300b5","observation_id":"dcbed4ea-62a8-48a1-b727-be0d3cb6004e","resolution":{"observed_at":"2026-07-03T10:37:56.593959Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2606.29815","last_updated":"2026-06-29T05:48:42Z","snapshot_observed_at":"2026-08-08T02:01:37.807950Z","submitted_at":"2026-06-29T05:48:42Z","title":"SrDetection: A Self-Referential Framework for Data Leakage Detection in Code Large Language Models","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-06-30T06:20:02.046020Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2606.29815"},"observation_digest":"sha256:4499d89ca6eb50ca96a4d4e86a99112d230d78621d321750e6552cb9fae2438b","observation_id":"df54faea-0bd2-4570-a54c-c347743948f7","resolution":{"observed_at":"2026-06-30T06:24:18.367339Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-01T05:03:42.877641Z","title":"arXiv preprint arXiv:2311.04850 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22368","last_updated":"2026-07-24T14:55:19Z","snapshot_observed_at":"2026-08-06T19:52:38.505220Z","submitted_at":"2026-07-24T14:55:19Z","title":"Do Agent Benchmarks Measure Capability? Protocol Validity in the Age of Agentic AI","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-01T05:03:42.877641Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2607.22368"},"observation_digest":"sha256:d73ca63694a1346de5b59e9887e7d7d3a85512eac999cbdb2fe0c976ed0d51e9","observation_id":"e34d5daa-ff82-4dcf-8c9b-19274d08c5d6","resolution":{"observed_at":"2026-08-01T05:03:42.877641Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-08T04:31:04.418737Z","title":"Gonzalez, and Ion Stoica","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.418737Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:e20c460f9083d065f5a9c962a71c81cd0a4fc2ddbdf7ab747f3cf150918ae6a1","observation_id":"372bad00-77da-4547-981f-6351c5a94418","resolution":{"observed_at":"2026-08-08T04:31:04.418737Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-06T22:23:05.614586Z","title":"Solve Rate","venue":null,"work_id":null,"year":2002},"citing_paper":{"arxiv_id":"2608.04549","last_updated":"2026-08-06T11:12:23Z","snapshot_observed_at":"2026-08-09T03:11:38.789617Z","submitted_at":"2026-08-05T07:39:49Z","title":"EuroExec: Frontier Language Models Fall Short of Expert Judgment on European Executive Decision Tasks","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-06T22:23:05.614586Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2608.04549"},"observation_digest":"sha256:cc8d225ae9c97db3da281feab6183c33b85aaa90f92bcb216dfa6dc94ad99289","observation_id":"f312928c-1d49-415a-bc1e-92f5c4a57e6c","resolution":{"observed_at":"2026-08-06T22:23:05.614586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-08T18:19:46.939354Z","title":"Solve Rate","venue":null,"work_id":null,"year":2002},"citing_paper":{"arxiv_id":"2608.04549","last_updated":"2026-08-06T11:12:23Z","snapshot_observed_at":"2026-08-09T03:11:38.789617Z","submitted_at":"2026-08-05T07:39:49Z","title":"EuroExec: Frontier Language Models Fall Short of Expert Judgment on European Executive Decision Tasks","version":2},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-08T18:19:46.939354Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2608.04549"},"observation_digest":"sha256:563ba61e670056600c17c723ed75434c567fd327ad1bcb4a5ebe8f0782048a89","observation_id":"1fd6bd95-accd-4fa4-b360-c9160579218f","resolution":{"observed_at":"2026-08-08T18:19:46.939354Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2311.04850/citation-record","integrity":"/paper/2311.04850/integrity","json":"/paper/2311.04850/citation-record.json","paper":"/paper/2311.04850"},"outbound":[],"paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 42 inbound Pith citation observations for arXiv:2311.04850."}