{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:643GD2UBUKDYLXOXV5IU6VRUSU","short_pith_number":"pith:643GD2UB","schema_version":"1.0","canonical_sha256":"f73661ea81a28785ddd7af514f5634953e893826ef2c8bccecbe3b798b312a7a","source":{"kind":"arxiv","id":"2501.01257","version":2},"attestation_state":"computed","paper":{"title":"CodeElo: Benchmarking Competition-level Code Generation of LLMs with Human-comparable Elo Ratings","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"An Yang, Binyuan Hui, Bofei Gao, Bowen Yu, Bo Zheng, Dayiheng Liu, Jian Yang, Jiaxi Yang, Junyang Lin, Shanghaoran Quan, Xuancheng Ren, Yang Fan, Yibo Miao, Yichang Zhang, Yunlong Feng, Zekun Wang, Zeyu Cui","submitted_at":"2025-01-02T13:49:00Z","abstract_excerpt":"With the increasing code reasoning capabilities of existing large language models (LLMs) and breakthroughs in reasoning models like OpenAI o1 and o3, there is a growing need to develop more challenging and comprehensive benchmarks that effectively test their sophisticated competition-level coding abilities. Existing benchmarks, like LiveCodeBench and USACO, fall short due to the unavailability of private test cases, lack of support for special judges, and misaligned execution environments. To bridge this gap, we introduce CodeElo, a standardized competition-level code generation benchmark that"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.01257","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-01-02T13:49:00Z","cross_cats_sorted":[],"title_canon_sha256":"e0b31860b29d5230af3d5e8ff59497b5465951c7976da87790dfe3ab43cf74e2","abstract_canon_sha256":"c0a20f9968dfd32aadff58b71d0921165d6e309dfda3cc37bff9ee28efb8408b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:56:39.221174Z","signature_b64":"ZzLRsjQEXliCPLoPcJERLF5oez3i/SWmSZPpauFRftI9igy/+k9n4LBWjis1bcU6NT3vrg/xP4FBtHJnoXXRBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f73661ea81a28785ddd7af514f5634953e893826ef2c8bccecbe3b798b312a7a","last_reissued_at":"2026-07-05T09:56:39.220669Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:56:39.220669Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CodeElo: Benchmarking Competition-level Code Generation of LLMs with Human-comparable Elo Ratings","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"An Yang, Binyuan Hui, Bofei Gao, Bowen Yu, Bo Zheng, Dayiheng Liu, Jian Yang, Jiaxi Yang, Junyang Lin, Shanghaoran Quan, Xuancheng Ren, Yang Fan, Yibo Miao, Yichang Zhang, Yunlong Feng, Zekun Wang, Zeyu Cui","submitted_at":"2025-01-02T13:49:00Z","abstract_excerpt":"With the increasing code reasoning capabilities of existing large language models (LLMs) and breakthroughs in reasoning models like OpenAI o1 and o3, there is a growing need to develop more challenging and comprehensive benchmarks that effectively test their sophisticated competition-level coding abilities. Existing benchmarks, like LiveCodeBench and USACO, fall short due to the unavailability of private test cases, lack of support for special judges, and misaligned execution environments. To bridge this gap, we introduce CodeElo, a standardized competition-level code generation benchmark that"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.01257","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.01257/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.01257","created_at":"2026-07-05T09:56:39.220729+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.01257v2","created_at":"2026-07-05T09:56:39.220729+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.01257","created_at":"2026-07-05T09:56:39.220729+00:00"},{"alias_kind":"pith_short_12","alias_value":"643GD2UBUKDY","created_at":"2026-07-05T09:56:39.220729+00:00"},{"alias_kind":"pith_short_16","alias_value":"643GD2UBUKDYLXOX","created_at":"2026-07-05T09:56:39.220729+00:00"},{"alias_kind":"pith_short_8","alias_value":"643GD2UB","created_at":"2026-07-05T09:56:39.220729+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":20,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.07748","citing_title":"Selective Left-Shift: Turning Test-Time Compute and Difficulty-based Curation into Training Data for Low-Resource Code Generation","ref_index":21,"is_internal_anchor":true},{"citing_arxiv_id":"2607.08009","citing_title":"From Execution to Education: A Bloom-Aligned Framework for Measuring Educational Control in LLMs","ref_index":145,"is_internal_anchor":true},{"citing_arxiv_id":"2606.25450","citing_title":"The Generalization Spectrum: A Chromatographic Approach to Evaluating Learning Algorithms","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2606.25450","citing_title":"The Generalization Spectrum: A Chromatographic Approach to Evaluating Learning Algorithms","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21228","citing_title":"Sakana Fugu Technical Report","ref_index":157,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00248","citing_title":"Seed2.0 Model Card: Towards Intelligence Frontier for Real-World Complexity","ref_index":89,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17333","citing_title":"Leveraging Error Diversity in Group Rollouts for Reinforcement Learning","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30394","citing_title":"CodeGolf Bench: A Multi-Language Benchmark for Evaluating Concise Code Generation Capabilities of Large Language Models","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07731","citing_title":"Benchmarking EngGPT2-16B-A3B against Comparable Italian and International Open-source LLMs","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17333","citing_title":"Leveraging Error Diversity in Group Rollouts for Reinforcement Learning","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15301","citing_title":"Solvita: Enhancing Large Language Models for Competitive Programming via Agentic Evolution","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2509.19349","citing_title":"ShinkaEvolve: Towards Open-Ended And Sample-Efficient Program Evolution","ref_index":121,"is_internal_anchor":false},{"citing_arxiv_id":"2506.11763","citing_title":"DeepResearch Bench: A Comprehensive Benchmark for Deep Research Agents","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02721","citing_title":"GrandCode: Achieving Grandmaster Level in Competitive Programming via Agentic Reinforcement Learning","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12384","citing_title":"Scalable Token-Level Hallucination Detection in Large Language Models","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08478","citing_title":"When Independent Sampling Outperforms Agentic Reasoning","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04779","citing_title":"A meta-analysis of the effect of generative AI on productivity and learning in programming","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07307","citing_title":"Rethinking Dense Sequential Chains: Reasoning Language Models Can Extract Answers from Sparse, Order-Shuffling Chain-of-Thoughts","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07731","citing_title":"Benchmarking EngGPT2-16B-A3B against Comparable Italian and International Open-source LLMs","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2505.09388","citing_title":"Qwen3 Technical Report","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/643GD2UBUKDYLXOXV5IU6VRUSU","json":"https://pith.science/pith/643GD2UBUKDYLXOXV5IU6VRUSU.json","graph_json":"https://pith.science/api/pith-number/643GD2UBUKDYLXOXV5IU6VRUSU/graph.json","events_json":"https://pith.science/api/pith-number/643GD2UBUKDYLXOXV5IU6VRUSU/events.json","paper":"https://pith.science/paper/643GD2UB"},"agent_actions":{"view_html":"https://pith.science/pith/643GD2UBUKDYLXOXV5IU6VRUSU","download_json":"https://pith.science/pith/643GD2UBUKDYLXOXV5IU6VRUSU.json","view_paper":"https://pith.science/paper/643GD2UB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.01257&json=true","fetch_graph":"https://pith.science/api/pith-number/643GD2UBUKDYLXOXV5IU6VRUSU/graph.json","fetch_events":"https://pith.science/api/pith-number/643GD2UBUKDYLXOXV5IU6VRUSU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/643GD2UBUKDYLXOXV5IU6VRUSU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/643GD2UBUKDYLXOXV5IU6VRUSU/action/storage_attestation","attest_author":"https://pith.science/pith/643GD2UBUKDYLXOXV5IU6VRUSU/action/author_attestation","sign_citation":"https://pith.science/pith/643GD2UBUKDYLXOXV5IU6VRUSU/action/citation_signature","submit_replication":"https://pith.science/pith/643GD2UBUKDYLXOXV5IU6VRUSU/action/replication_record"}},"created_at":"2026-07-05T09:56:39.220729+00:00","updated_at":"2026-07-05T09:56:39.220729+00:00"}