{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:B2W5U7HKGSIKHBWLX2HTEXX65R","short_pith_number":"pith:B2W5U7HK","schema_version":"1.0","canonical_sha256":"0eadda7cea3490a386cbbe8f325efeec4b5baef41cc93b7a280833fcd8253f23","source":{"kind":"arxiv","id":"2310.10628","version":1},"attestation_state":"computed","paper":{"title":"Data Contamination Through the Lens of Time","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Christine Herlihy, Colin White, Himanshu Thakur, Manley Roberts, Samuel Dooley","submitted_at":"2023-10-16T17:51:29Z","abstract_excerpt":"Recent claims about the impressive abilities of large language models (LLMs) are often supported by evaluating publicly available benchmarks. Since LLMs train on wide swaths of the internet, this practice raises concerns of data contamination, i.e., evaluating on examples that are explicitly or implicitly included in the training data. Data contamination remains notoriously challenging to measure and mitigate, even with partial attempts like controlled experimentation of training data, canary strings, or embedding similarities. In this work, we conduct the first thorough longitudinal analysis "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.10628","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-10-16T17:51:29Z","cross_cats_sorted":[],"title_canon_sha256":"27c50443c5d0cb5fefed547287ffdf2a957a027cce2ec94d907803f0d026886c","abstract_canon_sha256":"4fa197e15fd4d56a06a0548d5a93778ac1f324ab3022ee02a8ab87257a9cc43e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:01:22.584998Z","signature_b64":"F20f5AqgyIDGpCJecS9KQWzrTJwgz0Mix0CYFQnSEJA3QHe/cx1y4wqjJA4pA2+Qfi6rIOpQ4zgNdqiVyq2rDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0eadda7cea3490a386cbbe8f325efeec4b5baef41cc93b7a280833fcd8253f23","last_reissued_at":"2026-07-05T07:01:22.584509Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:01:22.584509Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Data Contamination Through the Lens of Time","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Christine Herlihy, Colin White, Himanshu Thakur, Manley Roberts, Samuel Dooley","submitted_at":"2023-10-16T17:51:29Z","abstract_excerpt":"Recent claims about the impressive abilities of large language models (LLMs) are often supported by evaluating publicly available benchmarks. Since LLMs train on wide swaths of the internet, this practice raises concerns of data contamination, i.e., evaluating on examples that are explicitly or implicitly included in the training data. Data contamination remains notoriously challenging to measure and mitigate, even with partial attempts like controlled experimentation of training data, canary strings, or embedding similarities. In this work, we conduct the first thorough longitudinal analysis "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.10628","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.10628/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.10628","created_at":"2026-07-05T07:01:22.584568+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.10628v1","created_at":"2026-07-05T07:01:22.584568+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.10628","created_at":"2026-07-05T07:01:22.584568+00:00"},{"alias_kind":"pith_short_12","alias_value":"B2W5U7HKGSIK","created_at":"2026-07-05T07:01:22.584568+00:00"},{"alias_kind":"pith_short_16","alias_value":"B2W5U7HKGSIKHBWL","created_at":"2026-07-05T07:01:22.584568+00:00"},{"alias_kind":"pith_short_8","alias_value":"B2W5U7HK","created_at":"2026-07-05T07:01:22.584568+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.23628","citing_title":"How Hard is it to Rig a Benchmark? A Social Choice Analysis of Leaderboard Robustness","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2406.04244","citing_title":"Benchmark Data Contamination of Large Language Models: A Survey","ref_index":126,"is_internal_anchor":false},{"citing_arxiv_id":"2403.07974","citing_title":"LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code","ref_index":207,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/B2W5U7HKGSIKHBWLX2HTEXX65R","json":"https://pith.science/pith/B2W5U7HKGSIKHBWLX2HTEXX65R.json","graph_json":"https://pith.science/api/pith-number/B2W5U7HKGSIKHBWLX2HTEXX65R/graph.json","events_json":"https://pith.science/api/pith-number/B2W5U7HKGSIKHBWLX2HTEXX65R/events.json","paper":"https://pith.science/paper/B2W5U7HK"},"agent_actions":{"view_html":"https://pith.science/pith/B2W5U7HKGSIKHBWLX2HTEXX65R","download_json":"https://pith.science/pith/B2W5U7HKGSIKHBWLX2HTEXX65R.json","view_paper":"https://pith.science/paper/B2W5U7HK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.10628&json=true","fetch_graph":"https://pith.science/api/pith-number/B2W5U7HKGSIKHBWLX2HTEXX65R/graph.json","fetch_events":"https://pith.science/api/pith-number/B2W5U7HKGSIKHBWLX2HTEXX65R/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/B2W5U7HKGSIKHBWLX2HTEXX65R/action/timestamp_anchor","attest_storage":"https://pith.science/pith/B2W5U7HKGSIKHBWLX2HTEXX65R/action/storage_attestation","attest_author":"https://pith.science/pith/B2W5U7HKGSIKHBWLX2HTEXX65R/action/author_attestation","sign_citation":"https://pith.science/pith/B2W5U7HKGSIKHBWLX2HTEXX65R/action/citation_signature","submit_replication":"https://pith.science/pith/B2W5U7HKGSIKHBWLX2HTEXX65R/action/replication_record"}},"created_at":"2026-07-05T07:01:22.584568+00:00","updated_at":"2026-07-05T07:01:22.584568+00:00"}