{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:OXQIXBIM5E3UQB76YMIMAM2NKY","short_pith_number":"pith:OXQIXBIM","schema_version":"1.0","canonical_sha256":"75e08b850ce9374807fec310c0334d560bc09aea3cf56496d4b2c2fb9a0c6649","source":{"kind":"arxiv","id":"2504.01995","version":2},"attestation_state":"computed","paper":{"title":"Brains vs. Bytes: Evaluating LLM Proficiency in Olympiad Mathematics","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Alireza Farhadi, Alireza Hashemi, Amir Khasahmadi, Hamed Mahdavi, Majid Daliri, Pegah Mohammadipour, Samira Malek, Vasant Honavar, Yekta Yazdanifard","submitted_at":"2025-04-01T00:10:10Z","abstract_excerpt":"Recent advances in large language models (LLMs) have shown impressive progress in mathematical reasoning tasks. However, current evaluation benchmarks predominantly focus on the accuracy of final answers, often overlooking the crucial logical rigor for mathematical problem solving. The claim that state-of-the-art LLMs can solve Math Olympiad-level problems requires closer examination. To explore this, we conducted both qualitative and quantitative human evaluations of proofs generated by LLMs, and developed a schema for automatically assessing their reasoning capabilities. Our study reveals th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.01995","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-04-01T00:10:10Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"b29c864d47b71051245d01257331d532e6a2cfcd9e2437257a154a385a4a0971","abstract_canon_sha256":"3db214dcb73bd2c78c783e01426ae4d80037e54df0ff9b6b972ec9bfd1c2ac00"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:47:37.844930Z","signature_b64":"xvyUQBTVr3pJdXwEMOsehoD5/zGGi6VYJLXP0P+EwLIco7atuaJiE1vde0TAkzGnpjx7erVjxLafA/IcBReRCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"75e08b850ce9374807fec310c0334d560bc09aea3cf56496d4b2c2fb9a0c6649","last_reissued_at":"2026-07-05T10:47:37.844469Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:47:37.844469Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Brains vs. Bytes: Evaluating LLM Proficiency in Olympiad Mathematics","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Alireza Farhadi, Alireza Hashemi, Amir Khasahmadi, Hamed Mahdavi, Majid Daliri, Pegah Mohammadipour, Samira Malek, Vasant Honavar, Yekta Yazdanifard","submitted_at":"2025-04-01T00:10:10Z","abstract_excerpt":"Recent advances in large language models (LLMs) have shown impressive progress in mathematical reasoning tasks. However, current evaluation benchmarks predominantly focus on the accuracy of final answers, often overlooking the crucial logical rigor for mathematical problem solving. The claim that state-of-the-art LLMs can solve Math Olympiad-level problems requires closer examination. To explore this, we conducted both qualitative and quantitative human evaluations of proofs generated by LLMs, and developed a schema for automatically assessing their reasoning capabilities. Our study reveals th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.01995","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.01995/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.01995","created_at":"2026-07-05T10:47:37.844518+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.01995v2","created_at":"2026-07-05T10:47:37.844518+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.01995","created_at":"2026-07-05T10:47:37.844518+00:00"},{"alias_kind":"pith_short_12","alias_value":"OXQIXBIM5E3U","created_at":"2026-07-05T10:47:37.844518+00:00"},{"alias_kind":"pith_short_16","alias_value":"OXQIXBIM5E3UQB76","created_at":"2026-07-05T10:47:37.844518+00:00"},{"alias_kind":"pith_short_8","alias_value":"OXQIXBIM","created_at":"2026-07-05T10:47:37.844518+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.10379","citing_title":"Not All Proofs Are Equal: Evaluating LLM Proof Quality Beyond Correctness","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19338","citing_title":"STAR-P\\'olyaMath: Multi-Agent Reasoning under Persistent Meta-Strategic Supervision","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2505.23281","citing_title":"MathArena: Evaluating LLMs on Uncontaminated Math Competitions","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10379","citing_title":"Not All Proofs Are Equal: Evaluating LLM Proof Quality Beyond Correctness","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OXQIXBIM5E3UQB76YMIMAM2NKY","json":"https://pith.science/pith/OXQIXBIM5E3UQB76YMIMAM2NKY.json","graph_json":"https://pith.science/api/pith-number/OXQIXBIM5E3UQB76YMIMAM2NKY/graph.json","events_json":"https://pith.science/api/pith-number/OXQIXBIM5E3UQB76YMIMAM2NKY/events.json","paper":"https://pith.science/paper/OXQIXBIM"},"agent_actions":{"view_html":"https://pith.science/pith/OXQIXBIM5E3UQB76YMIMAM2NKY","download_json":"https://pith.science/pith/OXQIXBIM5E3UQB76YMIMAM2NKY.json","view_paper":"https://pith.science/paper/OXQIXBIM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.01995&json=true","fetch_graph":"https://pith.science/api/pith-number/OXQIXBIM5E3UQB76YMIMAM2NKY/graph.json","fetch_events":"https://pith.science/api/pith-number/OXQIXBIM5E3UQB76YMIMAM2NKY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OXQIXBIM5E3UQB76YMIMAM2NKY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OXQIXBIM5E3UQB76YMIMAM2NKY/action/storage_attestation","attest_author":"https://pith.science/pith/OXQIXBIM5E3UQB76YMIMAM2NKY/action/author_attestation","sign_citation":"https://pith.science/pith/OXQIXBIM5E3UQB76YMIMAM2NKY/action/citation_signature","submit_replication":"https://pith.science/pith/OXQIXBIM5E3UQB76YMIMAM2NKY/action/replication_record"}},"created_at":"2026-07-05T10:47:37.844518+00:00","updated_at":"2026-07-05T10:47:37.844518+00:00"}