{"as_of":"2026-08-10T20:19:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:8f6640b4f808ea0e6d5f51aae0c466a84da0abbb6f8455ddee2d2399055fd724","coverage":[{"denominator":79,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":79,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T14:44:09.471855Z","state":"measured"},{"denominator":79,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":79,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2507.17747/citation-record","integrity":"/paper/2507.17747/integrity","json":"/paper/2507.17747/citation-record.json","paper":"/paper/2507.17747"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.907090Z","title":"Claude 3.5 sonnet","venue":null,"work_id":"39614098-eb34-40e7-91e9-d720b30d0a80","year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.233992Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:48d8bada4b33c62a05d37067a389564dfa5eba79e3e4fe43a06f48bb15cb30bb","observation_id":"a83d6f7c-cc0d-4c07-9845-7e11451947d1","resolution":{"observed_at":"2026-08-06T14:44:11.909879Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.898638Z","title":"Claude 3.5 haiku","venue":null,"work_id":"fdf37974-ef36-4f02-a750-35a5fc67dd4c","year":2025},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.237996Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:b8e34f104c00e6579d84f6166dc3409c0e576b7ed1157d449770faa3fcc74d16","observation_id":"0c351428-9b2f-4c72-bcd6-0b4c6cf93302","resolution":{"observed_at":"2026-08-06T14:44:11.901402Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.890035Z","title":"ARC-AGI-2 + ARC Prize 2025 is Live! Blog Post, https://arcprize.org/blog/announcing-arc-agi-2-and-arc-prize-2025, March 24 2025","venue":null,"work_id":"0081dad1-40b5-4b5b-94f5-e666ff04447d","year":2025},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.241713Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:5ee5b1271ca4b245838b1d0791aae8b3b5a5e9e41c6d8fb19dc96a793c3d3f58","observation_id":"00efc9cd-aaad-4095-8e28-dba4fd18b6a8","resolution":{"observed_at":"2026-08-06T14:44:11.893059Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.04181","last_updated":"2023-11-04T11:50:13Z","snapshot_observed_at":"2026-07-06T15:39:33.825891Z","submitted_at":"2023-06-07T06:29:58Z","title":"Benchmarking Foundation Models with Language-Model-as-an-Examiner","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.04181","snapshot_observed_at":"2026-08-06T14:44:09.245058Z","title":"Benchmarking foundation models with language-model-as-an-examiner","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.245058Z"},"links":{"cited_paper":"/paper/2306.04181","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:e2e2454488702eb71bac2d1429d859ad7b2487938a1846493e5837b1e0527e51","observation_id":"168020df-faa3-4617-bcb0-3f661ff0428e","resolution":{"observed_at":"2026-08-06T14:44:09.245058Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.881334Z","title":"Leak, cheat, repeat: Data contamination and evaluation malpractices in closed-source llms","venue":null,"work_id":"4c53be42-aea0-4a7a-8ef8-a33413de120b","year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.248532Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:1b1f7b2a621aae4ca7af93de19fd2e6a34c3d3f4fce537c77585e4b2eb614949","observation_id":"d2899a6c-dd33-44ef-9b7f-4fa173d23ace","resolution":{"observed_at":"2026-08-06T14:44:11.884260Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:09.252377Z","title":"Adversarial multi-agent evaluation of large language models through iterative debates","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.252377Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:16921ead90b935e278964c910e6dbcf1b1d22cb2ab49b844775fdd55124b8070","observation_id":"8e75f131-7bc5-43fa-b6b7-b20bf3ce0b61","resolution":{"observed_at":"2026-08-06T14:44:09.252377Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.872613Z","title":"Flageval","venue":null,"work_id":"9bc0cb3e-4ecc-4244-9cee-7778df2f4bda","year":2025},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.255787Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:12a93fde4106cc48a1fb138b5a82fee867ea238228040336bc5e02a827ece346","observation_id":"325cfa56-0309-4fc2-bdd2-eb2b2dc9dbb9","resolution":{"observed_at":"2026-08-06T14:44:11.875447Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13408","last_updated":"2025-05-19T17:44:26Z","snapshot_observed_at":"2026-08-07T15:43:00.197948Z","submitted_at":"2025-05-19T17:44:26Z","title":"CoT-Kinetics: A Theoretical Modeling Assessing LRM Reasoning Process","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.13408","snapshot_observed_at":"2026-08-06T14:44:09.258694Z","title":"Cot-kinetics: A theoretical modeling assessing lrm reasoning process, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.258694Z"},"links":{"cited_paper":"/paper/2505.13408","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:547608f27e832641e1be14dc1fe09f47b06f8b42eeed916f39cea9b0208ebf55","observation_id":"28e252be-bc05-477b-9b7c-6b1eb1892a54","resolution":{"observed_at":"2026-08-06T14:44:09.258694Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:09.261955Z","title":null,"venue":null,"work_id":null,"year":1952},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.261955Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:db68ab5210dc29f6ff7931e662020ec647650c16fd0d278566c1b4d9a4d57975","observation_id":"91128ddb-5cc4-4299-be72-ffd00b990fdd","resolution":{"observed_at":"2026-08-06T14:44:09.261955Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.12712","last_updated":"2023-04-13T20:41:31Z","snapshot_observed_at":"2026-08-03T04:49:15.195814Z","submitted_at":"2023-03-22T16:51:28Z","title":"Sparks of Artificial General Intelligence: Early experiments with GPT-4","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.12712","snapshot_observed_at":"2026-08-06T14:44:09.264731Z","title":"Sparks of artificial general intelligence: Early experiments with gpt-4, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.264731Z"},"links":{"cited_paper":"/paper/2303.12712","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:97a6fd21096aad426a0be1235f79bb9512ca529f964dac363a799ee9bf1af4a9","observation_id":"f00b4b03-06ee-4086-9adc-237c2aaeaa5e","resolution":{"observed_at":"2026-08-06T14:44:09.264731Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02892","last_updated":"2025-07-07T23:41:53Z","snapshot_observed_at":"2026-07-06T19:27:16.301809Z","submitted_at":"2024-10-03T18:30:47Z","title":"The Role of Deductive and Inductive Reasoning in Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02892","snapshot_observed_at":"2026-08-06T14:44:09.268106Z","title":"The role of deductive and inductive reasoning in large language models, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.268106Z"},"links":{"cited_paper":"/paper/2410.02892","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:67cb4d71f4b43a233551ae5e5270059d94ab27e7e6461c78d5f2ab443a6a6524","observation_id":"d8b5ddd1-7732-46aa-a3de-03b2b08f4d5c","resolution":{"observed_at":"2026-08-06T14:44:09.268106Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.863689Z","title":"Are we on the right way for evaluating large vision-language models? In A","venue":null,"work_id":"643f7f49-aa94-453d-b1ae-b0caf0356341","year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.271284Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:c4ae696bda2e30493cae70e802c4e7c364039431b65e4400221da29beeaf0395","observation_id":"9dca74ff-0d91-4e32-ae9c-e9c730ce254e","resolution":{"observed_at":"2026-08-06T14:44:11.866662Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.855413Z","title":"Jordan, Joseph E","venue":null,"work_id":"1d56a819-015d-471d-b8de-60984a907b2f","year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.274046Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:cea4e976603ff828cb7ff91e0609876c44ec5fb162182c12a3b030023f953a06","observation_id":"2000b724-3372-452c-975c-a70ce8d73640","resolution":{"observed_at":"2026-08-06T14:44:11.858241Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.04604","last_updated":"2025-01-08T05:24:50Z","snapshot_observed_at":"2026-08-04T14:31:04.378756Z","submitted_at":"2024-12-05T20:40:28Z","title":"ARC Prize 2024: Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.04604","snapshot_observed_at":"2026-08-06T14:44:09.277264Z","title":"Arc prize 2024: Technical report, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.277264Z"},"links":{"cited_paper":"/paper/2412.04604","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:b3445128775a0d9794fb26b725cb0adf29fc97c30ecc8a13f9aea117d35ee915","observation_id":"8877c7e4-7337-4901-9ce7-33441fd4fad0","resolution":{"observed_at":"2026-08-06T14:44:09.277264Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.846483Z","title":"Abstraction and reasoning corpus for artificial general intelligence (arc-agi), 2019","venue":null,"work_id":"75668d08-3ace-451b-9943-ec9cd5553266","year":2019},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.280482Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:603010a2b25da7e1315cb94152cf9bf023472f75d7948200be96b856df6a4a92","observation_id":"8e8b6024-00f3-460a-a420-2b2ccca90a02","resolution":{"observed_at":"2026-08-06T14:44:11.849645Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-07T01:45:38.840969Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-06T14:44:09.283458Z","title":"Training verifiers to solve math word problems","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.283458Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:af48761ed3b8cdbdfafe9e8e1d71261d8ff5a2b727e5d0b93a5e0f1304a2c186","observation_id":"85429f59-fd84-4ab2-a982-cdf409324331","resolution":{"observed_at":"2026-08-06T14:44:09.283458Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-08-10T17:02:50.054809Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-08-06T14:44:09.286882Z","title":"Deepseek-v3 technical report","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.286882Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:9c326615036617387a41f25012052cf34d9579a2e166215ca7415a670e965529","observation_id":"48812bb5-00e1-424c-b0e9-4dafced156f7","resolution":{"observed_at":"2026-08-06T14:44:09.286882Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-06T14:44:09.289691Z","title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.289691Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:53508fd5312c26a8eee99efed2c54253e42c00dfac56a88a8191c14be6597ff7","observation_id":"a3ba679b-47be-4e9a-b1ff-285680b552b7","resolution":{"observed_at":"2026-08-06T14:44:09.289691Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:09.292859Z","title":"Investigating data contamination in modern benchmarks for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.292859Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:ca0539f37f0552daf88a9c8cd1d07f9ca2284d7805afd9654565af36eb1e5272","observation_id":"ed6de936-dfbd-410d-b224-bb625d1ae89a","resolution":{"observed_at":"2026-08-06T14:44:09.292859Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.836932Z","title":"Stabilizing modality gap & lowering gradient norms improve zero-shot adversarial robustness of vlms","venue":null,"work_id":"5833d507-db24-4699-a81c-3064f07cf1de","year":2025},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.295646Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:ddc22e22be731d4273a1840050df0a87ab9cd1caab7c4b168dc078fd1509bd9f","observation_id":"a740c12f-d3a0-49c5-b8d7-ef683acdcc95","resolution":{"observed_at":"2026-08-06T14:44:11.840252Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.827180Z","title":"Improving zero-shot adversarial robustness in vision-language models by closed-form alignment of adversarial path simplices","venue":null,"work_id":"3c215420-7b39-409d-b7f2-e2d406928138","year":2025},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.298363Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:eb25439b7d06813477e42ac004bdcdc44cdfe605cab3b87a9149d839628c6d01","observation_id":"8f4b08d6-f95c-48ea-9b21-134bb8dd3aba","resolution":{"observed_at":"2026-08-06T14:44:11.830837Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.14325","last_updated":"2023-05-23T17:55:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-23T17:55:11Z","title":"Improving Factuality and Reasoning in Language Models through Multiagent Debate","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.14325","snapshot_observed_at":"2026-08-06T14:44:09.301684Z","title":"Tenenbaum, and Igor Mordatch","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.301684Z"},"links":{"cited_paper":"/paper/2305.14325","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:667f6000275339f2c24103dd9a8fef963058653a2b1dd2c44b5b8d26a800baaa","observation_id":"6bb92465-f91d-41de-9fff-5c67a3152038","resolution":{"observed_at":"2026-08-06T14:44:09.301684Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"gov/2010549","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:09.820323Z","title":null,"venue":null,"work_id":"886e5f58-f823-4c7d-b0fa-5871bd247c8c","year":2008},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.304648Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:d2cca82b75649ea9cb2277391203e943448932eace97084064f2c0ccf32bcc3c","observation_id":"a810d3f1-b2a7-4cbb-9524-d02ddea554e3","resolution":{"observed_at":"2026-08-06T14:44:09.824746Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.04872","last_updated":"2025-12-23T02:23:47Z","snapshot_observed_at":"2026-08-04T15:54:46.196160Z","submitted_at":"2024-11-07T17:07:35Z","title":"FrontierMath: A Benchmark for Evaluating Advanced Mathematical Reasoning in AI","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.04872","snapshot_observed_at":"2026-08-06T14:44:09.307985Z","title":"Frontiermath: A benchmark for evaluating advanced mathematical reasoning in ai, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.307985Z"},"links":{"cited_paper":"/paper/2411.04872","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:3c62870246d8083a7c3be28b5777d05690dc8992c9168170bf54324e88dc00ed","observation_id":"834de928-4ab6-4b41-a962-c2fad6bd8d13","resolution":{"observed_at":"2026-08-06T14:44:09.307985Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.818250Z","title":"Time travel in llms: Tracing data contamination in large language models","venue":null,"work_id":"1dbaed3e-3cce-49cb-b3b8-4abac6a63ea0","year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.311518Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:cd5403e6df52671d94ac4ab989bfa7377589eacd97d4612b073c700bd49b0bcc","observation_id":"0a8ff0aa-536d-4f47-a10f-ed347fddadcb","resolution":{"observed_at":"2026-08-06T14:44:11.821333Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-10T16:40:37.411115Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-06T14:44:09.314332Z","title":"The llama 3 herd of models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.314332Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:80f131f5c3d9cc97f115595e20666672a4793af9b74475a2571d9d3159596a25","observation_id":"29f26e3a-ff53-43ea-8d89-28c698a42493","resolution":{"observed_at":"2026-08-06T14:44:09.314332Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15594","last_updated":"2025-10-19T10:32:43Z","snapshot_observed_at":"2026-08-02T10:23:50.881300Z","submitted_at":"2024-11-23T16:03:35Z","title":"A Survey on LLM-as-a-Judge","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15594","snapshot_observed_at":"2026-08-06T14:44:09.317974Z","title":"A survey on llm-as-a-judge","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.317974Z"},"links":{"cited_paper":"/paper/2411.15594","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:2db68efddcd3e6cb045e2416e76bef76791e7799b5d4bc1078f39b0bd23dfbb1","observation_id":"3f84371c-551c-4272-9c46-4b5923fea23e","resolution":{"observed_at":"2026-08-06T14:44:09.317974Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.20245","last_updated":"2025-02-10T21:17:54Z","snapshot_observed_at":"2026-08-01T05:47:51.533388Z","submitted_at":"2024-10-26T18:21:44Z","title":"Improving Model Evaluation using SMART Filtering of Benchmark Datasets","version":2},"cited_work":{"arxiv_id":"2410.20245","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.20245","snapshot_observed_at":"2026-08-06T14:44:09.724072Z","title":"Improving Model Evaluation using SMART Filtering of Benchmark Datasets","venue":"cs.CL","work_id":"3f97366e-80a0-4e57-a83a-a721858127d0","year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.321034Z"},"links":{"cited_paper":"/paper/2410.20245","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:4cff725b305aa02d1d74f745150722330e1378b813386563a3e789fe47210073","observation_id":"52a345a1-6f49-40e1-810e-a7640175aa05","resolution":{"observed_at":"2026-08-06T14:44:09.727362Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.809614Z","title":"Measuring massive multitask language understanding","venue":null,"work_id":"34cb5b4b-77d3-42cf-bfd1-2bce3963f88f","year":2021},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.323991Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:645fc5c529ecb779923c115deef39da1517cd694b67198674cffad7b3e2ffa16","observation_id":"e85e20e6-37bf-4433-a5fb-1c42a9f3b091","resolution":{"observed_at":"2026-08-06T14:44:11.812461Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.799532Z","title":"Trueskill : A bayesian skill rating system","venue":null,"work_id":"67550a25-d637-4cd0-a495-7917b4c08486","year":2006},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.327209Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:41082b86da8947574ea2c084e30669697c2a35a79e5c7c9c13ddc0a7ad345c02","observation_id":"1af5e0d3-7f8a-4b50-b586-318c67eb5b8b","resolution":{"observed_at":"2026-08-06T14:44:11.803349Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:09.330561Z","title":"Lo RA : Low-rank adaptation of large language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.330561Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:0c1593d77e17a389de2e5ac0fb83390bb3b495cf813a6bcd3b0e9583c31f89c7","observation_id":"5126eeb1-cf26-4ab6-a34d-14f7327f25db","resolution":{"observed_at":"2026-08-06T14:44:09.330561Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1805.00899","last_updated":"2018-10-22T17:36:07Z","snapshot_observed_at":"2026-08-02T15:33:17.783178Z","submitted_at":"2018-05-02T16:27:32Z","title":"AI safety via debate","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1805.00899","snapshot_observed_at":"2026-08-06T14:44:09.333252Z","title":"Ai safety via debate","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.333252Z"},"links":{"cited_paper":"/paper/1805.00899","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:cf0a1163558770f72d88944060d99175337d6cd008ea5e85b142e01b2d87169b","observation_id":"47a8eb31-62ee-40b4-88d8-7030916e14a2","resolution":{"observed_at":"2026-08-06T14:44:09.333252Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06825","last_updated":"2023-10-10T17:54:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-10T17:54:58Z","title":"Mistral 7B","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06825","snapshot_observed_at":"2026-08-06T14:44:09.336865Z","title":"Jiang, Alexandre Sablayrolles, Arthur Mensch, et al","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.336865Z"},"links":{"cited_paper":"/paper/2310.06825","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:e767fd1e6e66392b2ba7af7027bc7e8ab0b42e6297ae849cf326ed823aca76a5","observation_id":"c9630232-1386-4a80-9503-c007f5e98ef4","resolution":{"observed_at":"2026-08-06T14:44:09.336865Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.04088","last_updated":"2024-01-08T18:47:34Z","snapshot_observed_at":"2026-08-08T06:16:25.839566Z","submitted_at":"2024-01-08T18:47:34Z","title":"Mixtral of Experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.04088","snapshot_observed_at":"2026-08-06T14:44:09.340177Z","title":"Jiang, Alexandre Sablayrolles, Antoine Roux, et al","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.340177Z"},"links":{"cited_paper":"/paper/2401.04088","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:bee176d1b4234984115b3eb86a2dc42e5e532da27d943389bda42d713ba79a54","observation_id":"b86f7a43-fbad-48c8-ab59-5f59d018b4c0","resolution":{"observed_at":"2026-08-06T14:44:09.340177Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.784244Z","title":"Bowman, Tim Rockt \\\"a schel, and Ethan Perez","venue":null,"work_id":"0efda67f-918e-4c8b-a548-398f95a548fc","year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.343125Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:6d86e52565490c264bfa5b4ad361fa804dd2d286975f69ec0b4c2097856a9d11","observation_id":"87e694a9-d61b-4e1a-be24-bfeeeeac2a80","resolution":{"observed_at":"2026-08-06T14:44:11.787423Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13124","last_updated":"2025-01-21T05:36:13Z","snapshot_observed_at":"2026-08-10T17:42:18.756153Z","submitted_at":"2025-01-21T05:36:13Z","title":"Debate Helps Weak-to-Strong Generalization","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13124","snapshot_observed_at":"2026-08-06T14:44:09.346312Z","title":"Debate helps weak-to-strong generalization","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.346312Z"},"links":{"cited_paper":"/paper/2501.13124","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:b3ba882ec1c6c1a6d4892ac475a91a9e15bccbefd5c252842b8c8592ce7b1ded","observation_id":"325bd2a9-eb0f-4a67-a6c4-a13267cefb10","resolution":{"observed_at":"2026-08-06T14:44:09.346312Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.05579","last_updated":"2024-12-10T05:49:12Z","snapshot_observed_at":"2026-07-31T01:42:39.468673Z","submitted_at":"2024-12-07T08:07:24Z","title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.05579","snapshot_observed_at":"2026-08-06T14:44:09.349366Z","title":"Llms-as-judges: A comprehensive survey on llm-based evaluation methods","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.349366Z"},"links":{"cited_paper":"/paper/2412.05579","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:813e50e4b4522f36a3db9576ea63a7b8dcb3a840399e1838fa791bbb28a8ba36","observation_id":"dcdbfae7-ea09-477a-93c7-932389ee3b53","resolution":{"observed_at":"2026-08-06T14:44:09.349366Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.09212","last_updated":"2024-01-17T19:09:57Z","snapshot_observed_at":"2026-08-04T18:52:19.083847Z","submitted_at":"2023-06-15T15:49:51Z","title":"CMMLU: Measuring massive multitask language understanding in Chinese","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.09212","snapshot_observed_at":"2026-08-06T14:44:09.352447Z","title":"Cmmlu: Measuring massive multitask language understanding in chinese","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.352447Z"},"links":{"cited_paper":"/paper/2306.09212","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:6f723fcd7b2181b9e3a0f7fa15fea100349573cb205374da2ca70b0022761924","observation_id":"88498f70-5335-422b-be70-d39ac2d85e91","resolution":{"observed_at":"2026-08-06T14:44:09.352447Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.19485","last_updated":"2024-10-25T11:41:27Z","snapshot_observed_at":"2026-07-06T19:39:38.201130Z","submitted_at":"2024-10-25T11:41:27Z","title":"A Debate-Driven Experiment on LLM Hallucinations and Accuracy","version":1},"cited_work":{"arxiv_id":"2410.19485","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.19485","snapshot_observed_at":"2026-08-06T14:44:09.655768Z","title":"A Debate-Driven Experiment on LLM Hallucinations and Accuracy","venue":"cs.CL","work_id":"9e695b87-a12e-4f66-a373-2e852c598c4c","year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.355365Z"},"links":{"cited_paper":"/paper/2410.19485","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:afec1c52a510124d7e554b0be085c8629051c02e07997a8cf9804516b9e8ded1","observation_id":"87452799-2260-4345-bd6e-15bd193c9117","resolution":{"observed_at":"2026-08-06T14:44:09.659401Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.774579Z","title":"Manning, Christopher R \\' e , Diana Acosta - Navas, Drew A","venue":null,"work_id":"1e3073a5-0ec7-44bd-94fd-8df3725548dc","year":2023},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.358457Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:5fdee097847e5f01573f63f2c4d576237bc77e211939a83585e61837827dfbf6","observation_id":"ed2720b2-bae9-47a1-b47c-1b33256a0864","resolution":{"observed_at":"2026-08-06T14:44:11.777855Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.733575Z","title":"Encouraging divergent thinking in large language models through multi-agent debate","venue":null,"work_id":"8e1ad76f-dfa8-4fd7-8f44-935fa3d1881e","year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.361156Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:f86b1ede117348a02bfcfe807325e87b8e049bcb014d7dff465272335ff24e64","observation_id":"b16022a5-2bc7-40ea-b7db-0dbdc8a78cd6","resolution":{"observed_at":"2026-08-06T14:44:11.754017Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.646538Z","title":"An empirical analysis on large language models in debate evaluation","venue":null,"work_id":"b0976f66-e2af-41df-8e3d-6c80d08c317b","year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.364110Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:5e70c5c30ecca1f5f7b6b1d51afaaadf2dc2179d1f468101c97ac41b06fceaa8","observation_id":"bcd0f28c-7357-451f-a019-43dd1ffd8c48","resolution":{"observed_at":"2026-08-06T14:44:11.699063Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.595088Z","title":"The Llama 4 herd: The beginning of a new era of natively multimodal AI innovation","venue":null,"work_id":"e2d5fb32-2a65-42f5-bfb3-6111e06e6f30","year":2025},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.367183Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:ade9dfeef4acb75ea4aeaa8e32e5c3fcb58e264029bf355e53b360936cf82137","observation_id":"c70ea09d-581d-4baf-a8ff-a4403bcf8c35","resolution":{"observed_at":"2026-08-06T14:44:11.638130Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2024.acl-long.341","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"Discursive socratic questioning: Evaluating the faithfulness of language models' understanding of discourse relations","venue":null,"work_id":"7f7f433d-4bae-426c-b772-71a143ceec84","year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.370593Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:bc45b68cbe4f0cebf5d6818c250c0a27bc6ae0f82b1e25cce34cd6821165ae6a","observation_id":"658269b3-bea3-41b0-9ee6-c5226b858a26","resolution":{"observed_at":"2026-08-06T14:44:09.550141Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01743","last_updated":"2025-03-07T09:05:58Z","snapshot_observed_at":"2026-08-09T11:59:10.408717Z","submitted_at":"2025-03-03T17:05:52Z","title":"Phi-4-Mini Technical Report: Compact yet Powerful Multimodal Language Models via Mixture-of-LoRAs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01743","snapshot_observed_at":"2026-08-06T14:44:09.373495Z","title":"Phi-4-Mini Technical Report: Compact yet Powerful Multimodal Language Models via Mixture-of-LoRAs , 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.373495Z"},"links":{"cited_paper":"/paper/2503.01743","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:ddcb2bd00cfffb1bb000114b6c94c76b02033f77b7f33f277f9696ca6786675b","observation_id":"963a0a2e-6217-4657-988a-4b2724e01846","resolution":{"observed_at":"2026-08-06T14:44:09.373495Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.336032Z","title":"Cheaper, better, faster, stronger","venue":null,"work_id":"c151b123-5f9b-4931-b203-da40d4fccb7d","year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.376438Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:804813c71442697695969110683112025705e28c06f66e270174177bd9b34df4","observation_id":"471f0037-bffb-4f85-927f-10aaadc82ad1","resolution":{"observed_at":"2026-08-06T14:44:11.454555Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.202204Z","title":"Mistral large","venue":null,"work_id":"505b3ec1-455e-4ca2-9130-2ad2069aac4a","year":2025},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.379133Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:84b55c412ebecaa9a5b2a5f8ea9a2e4bcd82e88cd581c8186ec641771708a914","observation_id":"48907f32-9cc1-47f7-a2bb-4fc3af06a2d2","resolution":{"observed_at":"2026-08-06T14:44:11.325353Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11044","last_updated":"2025-02-07T21:56:40Z","snapshot_observed_at":"2026-07-06T18:31:49.437336Z","submitted_at":"2024-06-16T19:02:31Z","title":"Evaluating the Performance of Large Language Models via Debates","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11044","snapshot_observed_at":"2026-08-06T14:44:09.381842Z","title":"Evaluating the performance of large language models via debates","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.381842Z"},"links":{"cited_paper":"/paper/2406.11044","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:664df003695bd4b434ccdf687f6b502c1d756fbbfd05e681f0268cb6f58c08dd","observation_id":"e966dac2-cd50-4d15-baaa-94f3454d0b18","resolution":{"observed_at":"2026-08-06T14:44:09.381842Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-06T14:44:09.384917Z","title":"Gpt-4 technical report, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.384917Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:9ad03709384d9bd4943ce6f3a0a5a4b8bf8f027a965ca2facee77057d58718ad","observation_id":"0933da57-b03d-4223-9983-e35edb8fc8f0","resolution":{"observed_at":"2026-08-06T14:44:09.384917Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-06T14:44:09.387932Z","title":"GPT-4o System Card","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.387932Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:242acd73052aeaef243f816d6e5d7f35c9d146c60c87420bfbcaab4f92955ae2","observation_id":"ec0c7b9f-1c02-404d-b2f5-9f6bb2b5342d","resolution":{"observed_at":"2026-08-06T14:44:09.387932Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:10.958064Z","title":"GPT-4o mini: advancing cost-efficient intelligence","venue":null,"work_id":"f921298a-6533-44b8-8fb2-cdbfaaf6d1da","year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.391394Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:e70bb292a26bc2571f0b71d46cef25e6cedbda4f75d5c17fd5bd773f839f4950","observation_id":"f817549b-8f37-4ac2-800d-c183768200b8","resolution":{"observed_at":"2026-08-06T14:44:11.063257Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.16720","last_updated":"2026-04-30T02:46:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-21T18:04:31Z","title":"OpenAI o1 System Card","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.16720","snapshot_observed_at":"2026-08-06T14:44:09.394084Z","title":"Openai o1 system card, 2024 c","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.394084Z"},"links":{"cited_paper":"/paper/2412.16720","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:1b469ef75fdcf558da18c3681b27365fcbd8417925fddf59e741a4b81c30826b","observation_id":"8c4d6f4f-37f0-4b11-a6fc-4272bd79a562","resolution":{"observed_at":"2026-08-06T14:44:09.394084Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:09.396998Z","title":"Chatterji, Faisal Ladhak, and Tatsunori Hashimoto","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.396998Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:5c3b62b5586e692421518f75bdc61811b22bbb33663fb0b06548d01a4b2bb868","observation_id":"81b8e3f6-eb82-4a21-94f3-71c9f31dfe3d","resolution":{"observed_at":"2026-08-06T14:44:09.396998Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:09.399824Z","title":"Mapping global dynamics of benchmark creation and saturation in artificial intelligence","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.399824Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:c99058699651091b9debce60d9ea70ffa2089c3fa21bede069301f88f478ff2a","observation_id":"067938fe-4300-44b8-b2e4-4beba9569ad5","resolution":{"observed_at":"2026-08-06T14:44:09.399824Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14249","last_updated":"2026-02-20T04:23:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-24T05:27:46Z","title":"Humanity's Last Exam","version":10},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.14249","snapshot_observed_at":"2026-08-06T14:44:09.402625Z","title":"Humanity's last exam","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.402625Z"},"links":{"cited_paper":"/paper/2501.14249","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:258d999aa382dde8b8e5be36bbbde70dab8b072f569e046402aba6e6d15f8df7","observation_id":"7053c87d-d62c-44aa-b800-fd743facd11e","resolution":{"observed_at":"2026-08-06T14:44:09.402625Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:10.699050Z","title":"Introducing gemini 2.0: our new ai model for the agentic era","venue":null,"work_id":"46745679-bebd-4496-9a14-2773208d0950","year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.405513Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:2ac3dd28285d8ca461e51c4852c31ade90d115c4ca39f4852c3ea7cf3780500e","observation_id":"f12d1973-fa38-4f70-9902-f1b6e486e20c","resolution":{"observed_at":"2026-08-06T14:44:10.854039Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:10.359908Z","title":"Multi-layered evaluation using a fusion of metrics and LLMs as judges in open-domain question answering","venue":null,"work_id":"fc5290ca-3f1b-433b-ab46-56b2badac6f4","year":2025},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.408457Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:2ced19e921749b4ad684858258877666ef0231db202be7beaf3f629f22108cb8","observation_id":"36bb7fd3-ed87-4e09-b768-ccd3d1c8dc91","resolution":{"observed_at":"2026-08-06T14:44:10.471219Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.12022","last_updated":"2023-11-20T18:57:34Z","snapshot_observed_at":"2026-08-10T12:02:35.919497Z","submitted_at":"2023-11-20T18:57:34Z","title":"GPQA: A Graduate-Level Google-Proof Q&A Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.12022","snapshot_observed_at":"2026-08-06T14:44:09.411020Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.411020Z"},"links":{"cited_paper":"/paper/2311.12022","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:48398f828f7ddcb35c39b987f646196fa6c53eff474ed501b26266272c25b7b5","observation_id":"07618cbf-e2a8-490c-bec3-fbc58bbe8bf3","resolution":{"observed_at":"2026-08-06T14:44:09.411020Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:09.413880Z","title":"Nlp evaluation in trouble: On the need to measure llm data contamination for each benchmark","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.413880Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:d69eff57638b6b0c04d7863565c2fdac8bed0bf78682be674955b4013dc42dbd","observation_id":"8bb4c05e-68cc-4951-855a-2f612619ae98","resolution":{"observed_at":"2026-08-06T14:44:09.413880Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.08632","last_updated":"2023-09-13T19:47:33Z","snapshot_observed_at":"2026-07-06T16:19:10.103540Z","submitted_at":"2023-09-13T19:47:33Z","title":"Pretraining on the Test Set Is All You Need","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.08632","snapshot_observed_at":"2026-08-06T14:44:09.416573Z","title":"Pretraining on the test set is all you need, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.416573Z"},"links":{"cited_paper":"/paper/2309.08632","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:d3b3b8e9c1bc2d975f65b7731f21cbd54a9a3ebdcf0a80d2643d784c9a634c01","observation_id":"3cf9e6df-393b-4b02-adde-6800a7912adf","resolution":{"observed_at":"2026-08-06T14:44:09.416573Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:10.294425Z","title":"Detecting pretraining data from large language models","venue":null,"work_id":"ed45d7cd-f59d-4609-8510-d3d53eab2bb0","year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.419654Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:41e7702367893efac1b205f1b95850a2e3dc23f496426b6f07033b3eb859a360","observation_id":"d5675478-26e6-415f-b752-bc3c18d2885d","resolution":{"observed_at":"2026-08-06T14:44:10.319102Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:10.247811Z","title":"Beyond the imitation game: Quantifying and extrapolating the capabilities of language models","venue":null,"work_id":"a00addbd-bb3e-4144-9b0e-d163b3ba8651","year":2023},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.422236Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:1d4c88d79c94fbd5dfcbb692b08b7a1d3d08412bc37c619bde2cafdf09b2a746","observation_id":"db9419df-7567-4932-9c0b-f3dd29fabb58","resolution":{"observed_at":"2026-08-06T14:44:10.271308Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.02257","last_updated":"2024-10-15T18:37:03Z","snapshot_observed_at":"2026-07-06T19:10:04.150678Z","submitted_at":"2024-09-03T19:31:03Z","title":"MMLU-Pro+: Evaluating Higher-Order Reasoning and Shortcut Learning in LLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.02257","snapshot_observed_at":"2026-08-06T14:44:09.424893Z","title":"MMLU-Pro+: Evaluating Higher-Order Reasoning and Shortcut Learning in LLMs","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.424893Z"},"links":{"cited_paper":"/paper/2409.02257","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:7f1c6e9a4aa6749302a8f13b15cbbf3522def64f8d6c38f9c9aa6cc8985c0849","observation_id":"9717d2a1-315a-4ac1-b51d-f7940e0fb2e5","resolution":{"observed_at":"2026-08-06T14:44:09.424893Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:10.180277Z","title":null,"venue":null,"work_id":"3445b39c-4ef7-44be-bc45-c5274d3f8bdb","year":2019},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.427865Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:55a1b554b63b65881a98a7f335b705f6abf3b1f2984881bab4b20f05795c14e6","observation_id":"1647dcfc-81fe-4705-a412-fd81c0ab1680","resolution":{"observed_at":"2026-08-06T14:44:10.208592Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:10.138234Z","title":null,"venue":null,"work_id":"cb6f3d19-ec92-4189-bd17-faa965bd78fe","year":2019},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.430445Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:cd60b397b1bb5a672f2674bd0d8f2ae93f3d67ef64cba255567eb7887eea8508","observation_id":"e7e1d3ac-fd53-430f-8ac1-ee4cd9cb21f5","resolution":{"observed_at":"2026-08-06T14:44:10.148317Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:09.433045Z","title":"Chi, Sharan Narang, Aakanksha Chowdhery, and Denny Zhou","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.433045Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:52c89b0f2df7575e604222984c60bb335acd6a96a7a9d641298ae578887d42bb","observation_id":"29e7c9eb-c1a0-4b40-a774-be1a12be7040","resolution":{"observed_at":"2026-08-06T14:44:09.433045Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:10.121451Z","title":"Mmlu-pro: A more robust and challenging multi-task language understanding benchmark","venue":null,"work_id":"5bfdf7da-abe0-4332-9d29-b567c191c161","year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.436267Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:5caacb37f35ee58abccc15f545971396e090248693819bb54d4b0438d37d8f02","observation_id":"afab2f8b-6539-46ef-83d0-73bada5393a3","resolution":{"observed_at":"2026-08-06T14:44:10.124690Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:10.111868Z","title":"Chain-of-thought prompting elicits reasoning in large language models","venue":null,"work_id":"e5209b90-1834-49a7-9e24-1dff850b2b9a","year":2022},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.439020Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:688cb962365aa5aa3a5bee4659594932168b274b3ad9d118bd543ce40e1d7c6e","observation_id":"ff3d821e-b5f0-47d2-aae8-11173c64161c","resolution":{"observed_at":"2026-08-06T14:44:10.115093Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:10.102582Z","title":"Livebench: A challenging, contamination-free LLM benchmark","venue":null,"work_id":"39c61bca-80e8-410f-83de-2b2d03ea8ac7","year":2025},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.441678Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:34f7c6ecb60f90d0a8fd47678ca68e346f77597ef794efd68711694718c33b44","observation_id":"6c16eb36-7f0b-4507-a872-929d4fc131ee","resolution":{"observed_at":"2026-08-06T14:44:10.105651Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2023.emnlp-main.325","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"QUD eval: The evaluation of questions under discussion discourse parsing","venue":null,"work_id":"3de6c37e-5c39-43c7-8ff5-6daae025714e","year":2023},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.444655Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:b9e019675d98c0397a7c03067bfa69b32dcb2b77e2926d72f635b1d50831619e","observation_id":"ff34c8d0-bf09-4d70-9d71-6d5258bfea53","resolution":{"observed_at":"2026-08-06T14:44:09.505113Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04244","last_updated":"2024-06-06T16:41:39Z","snapshot_observed_at":"2026-07-30T15:43:06.151242Z","submitted_at":"2024-06-06T16:41:39Z","title":"Benchmark Data Contamination of Large Language Models: A Survey","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04244","snapshot_observed_at":"2026-08-06T14:44:09.448188Z","title":"Benchmark data contamination of large language models: A survey, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.448188Z"},"links":{"cited_paper":"/paper/2406.04244","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:9b92221628d937e88f6ec49a3e003a011fb231bf144f90a1cff0dda20d239e25","observation_id":"f51edf2e-e3ea-4b73-b8f5-05d222f22c20","resolution":{"observed_at":"2026-08-06T14:44:09.448188Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15043","last_updated":"2024-06-03T06:02:39Z","snapshot_observed_at":"2026-08-10T01:47:48.554344Z","submitted_at":"2024-02-23T01:30:39Z","title":"KIEval: A Knowledge-grounded Interactive Evaluation Framework for Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.15043","snapshot_observed_at":"2026-08-06T14:44:09.451179Z","title":"Kieval: A knowledge-grounded interactive evaluation framework for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.451179Z"},"links":{"cited_paper":"/paper/2402.15043","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:d87750f378d66968f20acf5b9e4826c581fe55e6c2648e5acd39d7a9e4ea726c","observation_id":"faaed509-9aa0-41be-8f71-1d8c24b38fac","resolution":{"observed_at":"2026-08-06T14:44:09.451179Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20267","last_updated":"2024-10-07T02:53:44Z","snapshot_observed_at":"2026-08-09T10:08:33.613106Z","submitted_at":"2024-05-30T17:19:19Z","title":"Auto-Arena: Automating LLM Evaluations with Agent Peer Battles and Committee Discussions","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20267","snapshot_observed_at":"2026-08-06T14:44:09.454188Z","title":"Auto-arena: Automating llm evaluations with agent peer battles and committee discussions, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.454188Z"},"links":{"cited_paper":"/paper/2405.20267","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:f087445cf1b6dcd528c4266c67d0365a9d8580d8ad38de7746c3c3c6e1fb1abe","observation_id":"d1754a25-0780-4bf6-8e85-c1d0fa4eb958","resolution":{"observed_at":"2026-08-06T14:44:09.454188Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:10.092478Z","title":"Xing, Hao Zhang, Joseph E","venue":null,"work_id":"d3a9bf1b-79c3-4821-b941-488bf0324094","year":2023},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.457014Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:509b9271eb27f072f7416efbb448a7a0366495a0c48974014222277bf0b4b22b","observation_id":"28e1a61d-5759-49a8-a35d-c52eb1b52a35","resolution":{"observed_at":"2026-08-06T14:44:10.096336Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:10.082831Z","title":"Dyval: Dynamic evaluation of large language models for reasoning tasks","venue":null,"work_id":"6b7b406a-8385-435b-8c26-122c3cb48547","year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":75,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.459891Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:08dd9d27024200b180a9dcec770a4dc69e300ba14235b3e781a16fe401544696","observation_id":"aa7be3b1-c45d-452b-a424-1fc0ded6b452","resolution":{"observed_at":"2026-08-06T14:44:10.086077Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:09.462617Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":76,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.462617Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:167ae9e750eb381eb593e0a46b38e0ce9159c7dc1d673ba9fa1fc7fdde620d6a","observation_id":"7d0e033a-3e88-47c9-bea6-db99e9f6c28f","resolution":{"observed_at":"2026-08-06T14:44:09.462617Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:09.465929Z","title":"@esa (Ref","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.465929Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:e3ae775620c8edc8af1e4e18b21a961ed50d01726ddd3eb14449df2f6e588a2a","observation_id":"6edf6c22-fd91-48f8-8d29-35f64be39ffd","resolution":{"observed_at":"2026-08-06T14:44:09.465929Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:09.468994Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":78,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.468994Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:5e81f68499e70cf494a79418f4494a8762157440c02ad2d04f5892b81e1d343d","observation_id":"4339ae32-1247-4087-955b-68eaa972ae45","resolution":{"observed_at":"2026-08-06T14:44:09.468994Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:09.471855Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":79,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.471855Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:e9ec9405d7865024de7a0a492e126d3bf701429433324269b04d70a459f493dd","observation_id":"5fde41a7-2a87-47c6-9831-2c3a0ca7d3a2","resolution":{"observed_at":"2026-08-06T14:44:09.471855Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-09T15:27:11.483986Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks"},"reference_resolution":{"displayed":79,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":44,"verified_exact":5,"verified_fuzzy":30},"total_outbound_references":79},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 79 of 79 outbound references and 0 inbound Pith citation observations for arXiv:2507.17747."}