{"as_of":"2026-08-14T13:01:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:fdae780fb83927a76cf305c068b50d507f273239a030c81f5bbdc65e216c9573","coverage":[{"denominator":81,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":81,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T15:40:39.950474Z","state":"measured"},{"denominator":85,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":85,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-14T06:32:32.682623+00:00","state":"measured"},{"denominator":4,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":4,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-09T19:27:46.891540Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-02T22:17:25.670647Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13766","snapshot_observed_at":"2026-08-09T19:27:46.891540Z","title":"Ugmathbench: A diverse and dynamic benchmark for undergraduate-level mathematical reasoning with large language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2502.00334","last_updated":"2025-06-03T07:13:03Z","snapshot_observed_at":"2026-08-14T07:05:32.040329Z","submitted_at":"2025-02-01T06:42:02Z","title":"UGPhysics: A Comprehensive Benchmark for Undergraduate Physics Reasoning with Large Language Models","version":4},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-08-09T19:27:46.891540Z"},"links":{"cited_paper":"/paper/2501.13766","citing_paper":"/paper/2502.00334"},"observation_digest":"sha256:b1031c8fa9de427404acc46e177007d4287f60f3e996d0ff02cf0ffb0aac61a8","observation_id":"db9ebfaa-48ac-4194-918c-b6c7ac841939","resolution":{"observed_at":"2026-08-09T19:27:46.891540Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13766","snapshot_observed_at":"2026-08-07T10:44:59.476120Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.04592","last_updated":"2025-06-05T03:16:08Z","snapshot_observed_at":"2026-08-14T08:21:01.693223Z","submitted_at":"2025-06-05T03:16:08Z","title":"Safe: Enhancing Mathematical Reasoning in Large Language Models via Retrospective Step-aware Formal Verification","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-07T10:44:59.476120Z"},"links":{"cited_paper":"/paper/2501.13766","citing_paper":"/paper/2506.04592"},"observation_digest":"sha256:65e3ad7665ec4a87051c33e399fd32a40ba8d2462c9ac4ece022ecb1a10c3b21","observation_id":"2d14513f-748d-4736-b5d6-42ef87186ab8","resolution":{"observed_at":"2026-08-07T10:44:59.476120Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13766","snapshot_observed_at":"2026-08-04T10:36:25.330394Z","title":"Zhang, T","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2510.09517","last_updated":"2026-07-02T09:56:45Z","snapshot_observed_at":"2026-08-08T00:15:14.362197Z","submitted_at":"2025-10-10T16:28:43Z","title":"StatEval: A Comprehensive Benchmark for Large Language Models in Statistics","version":2},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-04T10:36:25.330394Z"},"links":{"cited_paper":"/paper/2501.13766","citing_paper":"/paper/2510.09517"},"observation_digest":"sha256:3510c1cba679aa8a806bd48f72071d147bd4ba272c659b04bdea4a49ba09f6e2","observation_id":"0a6fec3f-c75c-4d2a-aa25-1949f32e86ed","resolution":{"observed_at":"2026-08-04T10:36:25.330394Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"cited_work":{"arxiv_id":"2501.13766","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13766","snapshot_observed_at":"2026-07-02T22:17:25.670647Z","title":"arXiv preprint arXiv:2501.13766 , year=","venue":null,"work_id":"d9c1dad9-5b45-4ac4-be93-925bdfb2ca92","year":null},"citing_paper":{"arxiv_id":"2606.15079","last_updated":"2026-06-13T03:21:49Z","snapshot_observed_at":"2026-07-06T23:52:28.557859Z","submitted_at":"2026-06-13T03:21:49Z","title":"Ling and Ring 2.6 Technical Report: Efficient and Instant Agentic Intelligence at Trillion-Parameter Scale","version":1},"reference_index":191,"source":"arxiv_source","source_observed_at":"2026-07-02T22:10:59.568675Z"},"links":{"cited_paper":"/paper/2501.13766","citing_paper":"/paper/2606.15079"},"observation_digest":"sha256:af312b751699928e719e5592f171cd2c41f917090c47fd6cab3bef25cecef23d","observation_id":"e7eed9a8-b5c5-4b7a-8a56-6dc6792e5537","resolution":{"observed_at":"2026-07-02T22:17:25.672136Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2501.13766/citation-record","integrity":"/paper/2501.13766/integrity","json":"/paper/2501.13766/citation-record.json","paper":"/paper/2501.13766"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2402.00157","last_updated":"2024-09-16T19:20:59Z","snapshot_observed_at":"2026-08-13T04:29:45.408362Z","submitted_at":"2024-01-31T20:26:32Z","title":"Large Language Models for Mathematical Reasoning: Progresses and Challenges","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.00157","snapshot_observed_at":"2026-08-10T15:40:39.544581Z","title":"Large language models for mathematical reasoning: Progresses and challenges","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.544581Z"},"links":{"cited_paper":"/paper/2402.00157","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:20d17f5888b8d0efd45265fa4404986fe08a1073effcc159acb24ed852114cf4","observation_id":"112c66d7-04a1-4d1b-a21a-e7d108d90a83","resolution":{"observed_at":"2026-08-10T15:40:39.544581Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:40:39.550749Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.550749Z"},"links":{"citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:bce3b7c1445b8df0a7ed930c6cdd4a0c48d70cdae6754af16d8a66ac5a12d45d","observation_id":"53e3c48f-fd35-40f4-b26c-8fc824980633","resolution":{"observed_at":"2026-08-10T15:40:39.550749Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:40:39.556201Z","title":"Llama 3 model card","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.556201Z"},"links":{"citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:78b59aa43b23cdccbe5c41605c179d2f0d57d13a0e29ea1282d738d4d7bc5414","observation_id":"c91449ec-0e50-4a9c-b3b5-43239d8bd634","resolution":{"observed_at":"2026-08-10T15:40:39.556201Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1905.13319","last_updated":"2019-05-30T21:28:12Z","snapshot_observed_at":"2026-08-10T18:39:13.771066Z","submitted_at":"2019-05-30T21:28:12Z","title":"MathQA: Towards Interpretable Math Word Problem Solving with Operation-Based Formalisms","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1905.13319","snapshot_observed_at":"2026-08-10T15:40:39.561212Z","title":"Mathqa: Towards interpretable math word problem solving with operation-based formalisms","venue":null,"work_id":null,"year":1905},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.561212Z"},"links":{"cited_paper":"/paper/1905.13319","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:bb4e73a044864c95bb501f8977e11a9b48cfc8a3703de5687ff64c9419f7ad48","observation_id":"04d744b4-b66d-4209-832e-ef129bbd370c","resolution":{"observed_at":"2026-08-10T15:40:39.561212Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:40:41.302005Z","title":"Claude 3 family","venue":null,"work_id":"191e8372-a77d-48d6-b4ae-7046aeec97ef","year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.566715Z"},"links":{"citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:f3a94de07dc5ad5bb296eb4b821ba533df6947e9309f5f112d9b14240846a86a","observation_id":"cedec0c5-2d90-4b56-97d7-d3461e4a3b75","resolution":{"observed_at":"2026-08-10T15:40:41.307721Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10631","last_updated":"2024-03-15T19:14:39Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-16T17:54:07Z","title":"Llemma: An Open Language Model For Mathematics","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10631","snapshot_observed_at":"2026-08-10T15:40:39.572302Z","title":"Llemma: An open language model for mathematics","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.572302Z"},"links":{"cited_paper":"/paper/2310.10631","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:c453d71eda2412f1cf71d54a909340a193cf183a1c27e36064a3413005cd9b1c","observation_id":"83950808-6d63-42db-b4f7-8fd3b7983326","resolution":{"observed_at":"2026-08-10T15:40:39.572302Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:40:39.577730Z","title":"Numinamath 7b cot","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.577730Z"},"links":{"citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:87e4f49d581ff649f7e1f176da3698f05b76001afcdcfe7fed6ec71b7bbc59cc","observation_id":"b537a3de-8faf-455e-836c-f489bb6c4a66","resolution":{"observed_at":"2026-08-10T15:40:39.577730Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:40:41.275963Z","title":"Natural language input for a computer problem solving system","venue":null,"work_id":"a2c4b6cc-4a0a-49fd-847e-0357ce1ea2d2","year":1964},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.584189Z"},"links":{"citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:238d3383933941cb1ab4fc43c48e114f410244e8223d4835865e5b6bc450735d","observation_id":"8f23a55c-7aeb-46a0-91cc-5793641ba7ad","resolution":{"observed_at":"2026-08-10T15:40:41.281044Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:40:39.589484Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.589484Z"},"links":{"citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:7877a76a30f6594969e72ce170252546fa5f34874bacc6c15e85a9d6cf748774","observation_id":"0c29df4c-4659-4dda-bfc5-48162fea5bca","resolution":{"observed_at":"2026-08-10T15:40:39.589484Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2105.14517","last_updated":"2022-01-11T03:50:31Z","snapshot_observed_at":"2026-08-13T19:13:48.274752Z","submitted_at":"2021-05-30T12:34:17Z","title":"GeoQA: A Geometric Question Answering Benchmark Towards Multimodal Numerical Reasoning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2105.14517","snapshot_observed_at":"2026-08-10T15:40:39.594235Z","title":"Geoqa: A geometric question answering benchmark towards multimodal numerical reasoning","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.594235Z"},"links":{"cited_paper":"/paper/2105.14517","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:a82bbccb5fd345961bcb27089dd2de2bab62e1a576b770165bf039a4646e4767","observation_id":"2c239a0b-3c6a-4e05-9ba8-de94667fed2e","resolution":{"observed_at":"2026-08-10T15:40:39.594235Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2211.12588","last_updated":"2023-10-23T01:27:38Z","snapshot_observed_at":"2026-08-02T13:06:11.850456Z","submitted_at":"2022-11-22T21:06:00Z","title":"Program of Thoughts Prompting: Disentangling Computation from Reasoning for Numerical Reasoning Tasks","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2211.12588","snapshot_observed_at":"2026-08-10T15:40:39.599408Z","title":"Program of thoughts prompting: Disentangling computation from reasoning for numerical reasoning tasks","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.599408Z"},"links":{"cited_paper":"/paper/2211.12588","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:a7721cc3c4ae4fed0bfaf3a0436338677906538b1d106a43466f4ad5e568051a","observation_id":"b1c5f3d2-255d-4d38-9d90-b5c2442894f2","resolution":{"observed_at":"2026-08-10T15:40:39.599408Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:40:41.248195Z","title":"Theoremqa: A theorem-driven question answering dataset","venue":null,"work_id":"26c3de0c-87f9-4ac8-b465-e24930eac1d8","year":2023},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.604841Z"},"links":{"citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:f10ffd43a1b68e79af28092b1ca2e57ee6a76441eb6090dcca4d9de3231d3421","observation_id":"9ffa50c7-3651-4513-b136-2d80c55f9f43","resolution":{"observed_at":"2026-08-10T15:40:41.253873Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.08939","last_updated":"2024-05-28T04:32:09Z","snapshot_observed_at":"2026-08-13T04:19:21.806858Z","submitted_at":"2024-02-14T04:50:18Z","title":"Premise Order Matters in Reasoning with Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.08939","snapshot_observed_at":"2026-08-10T15:40:39.609708Z","title":"Premise order matters in reasoning with large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.609708Z"},"links":{"cited_paper":"/paper/2402.08939","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:caf177fe1c4561e7a358642320de01e0e2df604b80bb78aa5939cecb92a4da9a","observation_id":"3d0553c2-64d4-40b2-981f-03f6f2cacf01","resolution":{"observed_at":"2026-08-10T15:40:39.609708Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-14T02:43:01.480086Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-10T15:40:39.614608Z","title":"Training verifiers to solve math word problems","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.614608Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:1544a623450dfdb61f1d8a30decf17e3b6718273b6599ec3c2b08c4cd41d284d","observation_id":"4b5d623a-009c-45bf-98fc-8027e215b48c","resolution":{"observed_at":"2026-08-10T15:40:39.614608Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:40:41.230682Z","title":"Evaluating language models for mathematics through interactions","venue":null,"work_id":"579f67ab-6ed7-4873-82f1-a360c9a140d7","year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.619641Z"},"links":{"citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:85edd7f2122ae679fd075501e4b180d259a486ce1be17592d55ed2e8e2bb2e4d","observation_id":"00d1b62a-186b-4e91-9f97-523cf32dc4fb","resolution":{"observed_at":"2026-08-10T15:40:41.236541Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06066","last_updated":"2024-01-11T17:31:42Z","snapshot_observed_at":"2026-08-13T15:18:04.411100Z","submitted_at":"2024-01-11T17:31:42Z","title":"DeepSeekMoE: Towards Ultimate Expert Specialization in Mixture-of-Experts Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06066","snapshot_observed_at":"2026-08-10T15:40:39.624353Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.624353Z"},"links":{"cited_paper":"/paper/2401.06066","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:b569f2419386996de616169b099388996d054230bbb29534ae76ecbb4d18481c","observation_id":"b56fc7bb-e869-42d5-8e8c-bac6c4a40447","resolution":{"observed_at":"2026-08-10T15:40:39.624353Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:40:39.629739Z","title":"Deepseek-v2: A strong, economical, and efficient mixture-of-experts language model, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.629739Z"},"links":{"citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:ceef7f1cba0afd50ee42b1cad6bfe9b0bf73cce80222a93afc4aeaac658ddb7e","observation_id":"2ffca2f8-04ee-4faf-80dd-38c1b1d26ceb","resolution":{"observed_at":"2026-08-10T15:40:39.629739Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.09783","last_updated":"2024-04-03T23:29:03Z","snapshot_observed_at":"2026-08-13T05:23:58.579496Z","submitted_at":"2023-11-16T11:03:04Z","title":"Investigating Data Contamination in Modern Benchmarks for Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.09783","snapshot_observed_at":"2026-08-10T15:40:39.634533Z","title":"Investigating data contamination in modern benchmarks for large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.634533Z"},"links":{"cited_paper":"/paper/2311.09783","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:4f4a798f1d9a079e0ba0585d119fdc457910520ec7c59b67b062233d78634eec","observation_id":"02230e92-85ab-4b48-af74-053c66d8d60d","resolution":{"observed_at":"2026-08-10T15:40:39.634533Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15938","last_updated":"2024-05-31T17:49:03Z","snapshot_observed_at":"2026-08-13T04:10:47.901686Z","submitted_at":"2024-02-24T23:54:41Z","title":"Generalization or Memorization: Data Contamination and Trustworthy Evaluation for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.15938","snapshot_observed_at":"2026-08-10T15:40:39.639354Z","title":"Generalization or memorization: Data contamination and trustworthy evaluation for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.639354Z"},"links":{"cited_paper":"/paper/2402.15938","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:dc26a89d25af5e57d5c65712c6e5c70bf85fcad87ad474fa5128c4469f03bc2f","observation_id":"c3126efb-dc84-4429-b7ff-800f2f3c8d47","resolution":{"observed_at":"2026-08-10T15:40:39.639354Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.08493","last_updated":"2024-02-21T22:02:26Z","snapshot_observed_at":"2026-08-13T10:33:32.397439Z","submitted_at":"2023-08-16T16:48:57Z","title":"Time Travel in LLMs: Tracing Data Contamination in Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.08493","snapshot_observed_at":"2026-08-10T15:40:39.644278Z","title":"Time travel in llms: Tracing data contamination in large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.644278Z"},"links":{"cited_paper":"/paper/2308.08493","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:76c4f96359785b36081621ea357d0e759e529212259419c76fd9c96bae0b082c","observation_id":"077e38fe-1be2-4295-a652-7ca1a967eb0a","resolution":{"observed_at":"2026-08-10T15:40:39.644278Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.17452","last_updated":"2024-02-21T12:59:22Z","snapshot_observed_at":"2026-08-13T21:12:04.800343Z","submitted_at":"2023-09-29T17:59:38Z","title":"ToRA: A Tool-Integrated Reasoning Agent for Mathematical Problem Solving","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.17452","snapshot_observed_at":"2026-08-10T15:40:39.649051Z","title":"Tora: A tool-integrated reasoning agent for mathematical problem solving","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.649051Z"},"links":{"cited_paper":"/paper/2309.17452","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:fa6ae35467ea20c583589fa626f7a250c1a1114ce9e83785e144bf2864188280","observation_id":"eabd3cea-40ad-42db-a1e3-f89c85a61189","resolution":{"observed_at":"2026-08-10T15:40:39.649051Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14008","last_updated":"2024-06-06T13:19:44Z","snapshot_observed_at":"2026-08-03T03:39:09.398343Z","submitted_at":"2024-02-21T18:49:26Z","title":"OlympiadBench: A Challenging Benchmark for Promoting AGI with Olympiad-Level Bilingual Multimodal Scientific Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.14008","snapshot_observed_at":"2026-08-10T15:40:39.653756Z","title":"Olympiadbench: A challenging benchmark for promoting agi with olympiad-level bilingual multimodal scientific problems","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.653756Z"},"links":{"cited_paper":"/paper/2402.14008","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:019a28f4437917b814cb2f1f47f0dcd8f620c8f3935870e381414313ba630d7c","observation_id":"88227e78-1063-4d19-8980-340995b03504","resolution":{"observed_at":"2026-08-10T15:40:39.653756Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.14011","last_updated":"2024-05-08T07:34:06Z","snapshot_observed_at":"2026-08-13T04:34:59.203693Z","submitted_at":"2024-01-25T08:22:10Z","title":"CMMU: A Benchmark for Chinese Multi-modal Multi-type Question Understanding and Reasoning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.14011","snapshot_observed_at":"2026-08-10T15:40:39.658649Z","title":"Cmmu: A benchmark for chinese multi-modal multi-type question understanding and reasoning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.658649Z"},"links":{"cited_paper":"/paper/2401.14011","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:8538bf65b9e72e9b507d35dddc4b785cf58488b542a6e00cd1aa0382bc834daf","observation_id":"5880b332-0ed5-4872-b79c-90421489126f","resolution":{"observed_at":"2026-08-10T15:40:39.658649Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2009.03300","last_updated":"2021-01-12T18:57:11Z","snapshot_observed_at":"2026-08-13T20:44:28.824685Z","submitted_at":"2020-09-07T17:59:25Z","title":"Measuring Massive Multitask Language Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.03300","snapshot_observed_at":"2026-08-10T15:40:39.663637Z","title":"Measuring massive multitask language understanding","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.663637Z"},"links":{"cited_paper":"/paper/2009.03300","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:a7a89b4c042f1b1928d6b224a8649905ff218a104dfb6304667061ba553824c2","observation_id":"f6f556c5-0b3b-4e02-8c07-d2fca9794773","resolution":{"observed_at":"2026-08-10T15:40:39.663637Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2103.03874","last_updated":"2021-11-08T21:30:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-03-05T18:59:39Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2103.03874","snapshot_observed_at":"2026-08-10T15:40:39.668864Z","title":"Measuring mathematical problem solving with the math dataset","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.668864Z"},"links":{"cited_paper":"/paper/2103.03874","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:a4b3065fbe13bb8d695dfde6c443feaf28fdf53280be4a2f722bbdcbead3de01","observation_id":"c65a5993-c6da-4384-8b0e-fe79d8886ba3","resolution":{"observed_at":"2026-08-10T15:40:39.668864Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12753","last_updated":"2025-03-06T12:55:25Z","snapshot_observed_at":"2026-08-12T23:39:56.641372Z","submitted_at":"2024-06-18T16:20:53Z","title":"OlympicArena: Benchmarking Multi-discipline Cognitive Reasoning for Superintelligent AI","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12753","snapshot_observed_at":"2026-08-10T15:40:39.673875Z","title":"Olympicarena: Benchmarking multi-discipline cognitive reasoning for superintelligent ai","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.673875Z"},"links":{"cited_paper":"/paper/2406.12753","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:d51374a9b07d3b73de9aa5b6cc0982f515b427d82feaf1e6ad7eb68ab80ad770","observation_id":"b3d0c280-304c-49a7-a989-ab3e5fc7b561","resolution":{"observed_at":"2026-08-10T15:40:39.673875Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06825","last_updated":"2023-10-10T17:54:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-10T17:54:58Z","title":"Mistral 7B","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06825","snapshot_observed_at":"2026-08-10T15:40:39.678599Z","title":"Mistral 7b","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.678599Z"},"links":{"cited_paper":"/paper/2310.06825","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:b8e979a591583c688bf7ed73ece417722070186b2b3337605184c957490d9309","observation_id":"f8781234-6a3c-4196-bc36-f758951902b5","resolution":{"observed_at":"2026-08-10T15:40:39.678599Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06059","last_updated":"2024-01-11T17:24:49Z","snapshot_observed_at":"2026-08-13T04:44:25.401800Z","submitted_at":"2024-01-11T17:24:49Z","title":"Investigating Data Contamination for Pre-training Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06059","snapshot_observed_at":"2026-08-10T15:40:39.683641Z","title":"Investigating data contamination for pre-training language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.683641Z"},"links":{"cited_paper":"/paper/2401.06059","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:8f3d8942ac98b460fbe070924fe98b955071971f19bc4ef5d29dc1d1e420290d","observation_id":"9db68948-0a87-4acf-9eee-84d5d90e3d3f","resolution":{"observed_at":"2026-08-10T15:40:39.683641Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:40:39.688524Z","title":"Large language models are zero-shot reasoners","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.688524Z"},"links":{"citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:228908dfd0475d509567e30d29ea11fd4057f9b0b9cc5e21407c0acd0d681ba5","observation_id":"9078c86e-1a07-4823-8cd8-d942510a60f6","resolution":{"observed_at":"2026-08-10T15:40:39.688524Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:40:41.190322Z","title":"Mawps: A math word problem repository","venue":null,"work_id":"c37ea624-f7fd-417e-9b46-277baa56d5ae","year":2016},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.693132Z"},"links":{"citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:21aee394a2a015f9bac73e4dfa1a0c652182a17098ee98fcdafd255a088b2469","observation_id":"0dfad2ef-30b3-40ce-98e1-586cb125aa77","resolution":{"observed_at":"2026-08-10T15:40:41.195779Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:40:39.697680Z","title":"Solving quantitative reasoning problems with language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.697680Z"},"links":{"citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:96b594b6ec2ddb31820ff86ad21ab1e2ec63147a91bfdbcd2feffcbe58364e64","observation_id":"74f166fb-caf5-4f6e-86cc-e6d481253606","resolution":{"observed_at":"2026-08-10T15:40:39.697680Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04706","last_updated":"2024-03-07T18:00:40Z","snapshot_observed_at":"2026-08-13T00:59:53.557863Z","submitted_at":"2024-03-07T18:00:40Z","title":"Common 7B Language Models Already Possess Strong Math Capabilities","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04706","snapshot_observed_at":"2026-08-10T15:40:39.702855Z","title":"Common 7b language models already possess strong math capabilities","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.702855Z"},"links":{"cited_paper":"/paper/2403.04706","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:61ac13dd1cd2bedd0c9ab7744b7598757dbb0d38e8c36c3c335a42832332fe35","observation_id":"c921ab3e-8305-40ce-ac3d-f7a0b99bd94f","resolution":{"observed_at":"2026-08-10T15:40:39.702855Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.19255","last_updated":"2024-07-02T03:46:03Z","snapshot_observed_at":"2026-08-13T04:07:08.803193Z","submitted_at":"2024-02-29T15:26:14Z","title":"GSM-Plus: A Comprehensive Benchmark for Evaluating the Robustness of LLMs as Mathematical Problem Solvers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.19255","snapshot_observed_at":"2026-08-10T15:40:39.707901Z","title":"Gsm-plus: A comprehensive benchmark for evaluating the robustness of llms as mathematical problem solvers","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.707901Z"},"links":{"cited_paper":"/paper/2402.19255","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:f9dac12c707f9b01b27da8f5fed19a4c71b74b6b5253150f5667529d2cb707c8","observation_id":"3aa96eeb-b20d-4d50-bdeb-84b6ff0231e9","resolution":{"observed_at":"2026-08-10T15:40:39.707901Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1705.04146","last_updated":"2017-10-23T16:45:03Z","snapshot_observed_at":"2026-08-13T10:15:11.378091Z","submitted_at":"2017-05-11T13:04:47Z","title":"Program Induction by Rationale Generation : Learning to Solve and Explain Algebraic Word Problems","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1705.04146","snapshot_observed_at":"2026-08-10T15:40:39.712585Z","title":"Program induction by rationale generation: Learning to solve and explain algebraic word problems","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.712585Z"},"links":{"cited_paper":"/paper/1705.04146","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:d94d50ee6297d90abcc93d57ef7b9de04850d91737dd13832d35c41df5024f51","observation_id":"433d176f-0b6d-451d-8053-e732d9c6b62f","resolution":{"observed_at":"2026-08-10T15:40:39.712585Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12209","last_updated":"2024-05-20T17:52:29Z","snapshot_observed_at":"2026-08-13T15:51:50.035268Z","submitted_at":"2024-05-20T17:52:29Z","title":"MathBench: Evaluating the Theory and Application Proficiency of LLMs with a Hierarchical Mathematics Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.12209","snapshot_observed_at":"2026-08-10T15:40:39.717433Z","title":"Mathbench: Evaluating the theory and application proficiency of llms with a hierarchical mathematics benchmark","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.717433Z"},"links":{"cited_paper":"/paper/2405.12209","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:9f135c8d251a32951e7f3301d3dd7d76eb1dfb39cf558f9b99383868c06fe0ba","observation_id":"37d24616-1827-498e-a9f2-0b7911c0c595","resolution":{"observed_at":"2026-08-10T15:40:39.717433Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02255","last_updated":"2024-01-21T03:47:06Z","snapshot_observed_at":"2026-07-06T16:27:15.027202Z","submitted_at":"2023-10-03T17:57:24Z","title":"MathVista: Evaluating Mathematical Reasoning of Foundation Models in Visual Contexts","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02255","snapshot_observed_at":"2026-08-10T15:40:39.722180Z","title":"Mathvista: Evaluating mathematical reasoning of foundation models in visual contexts","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.722180Z"},"links":{"cited_paper":"/paper/2310.02255","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:5f8ffca711c828fccb0daf87d42c6d1eca759c53cb80c610455cdcdec12ab249","observation_id":"ccbf91a8-5b0e-41c4-9b04-f03f8f0a8a8c","resolution":{"observed_at":"2026-08-10T15:40:39.722180Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.08786","last_updated":"2022-03-03T12:10:58Z","snapshot_observed_at":"2026-07-06T11:01:05.577957Z","submitted_at":"2021-04-18T09:29:16Z","title":"Fantastically Ordered Prompts and Where to Find Them: Overcoming Few-Shot Prompt Order Sensitivity","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.08786","snapshot_observed_at":"2026-08-10T15:40:39.727254Z","title":"Fantastically ordered prompts and where to find them: Overcoming few-shot prompt order sensitivity","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.727254Z"},"links":{"cited_paper":"/paper/2104.08786","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:6dc26719114f6b4a74289affa9f571034132789d967a2c8d98c97e3cb449c2bb","observation_id":"14c2d6a3-7291-4e88-ade2-344cbdc376f4","resolution":{"observed_at":"2026-08-10T15:40:39.727254Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:40:41.164561Z","title":"Fairness-guided few-shot prompting for large language models","venue":null,"work_id":"9966c9c6-9ff4-47fc-ae4e-2434b4700d1f","year":2023},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.732105Z"},"links":{"citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:efe20b67c9a4af4238bc5b95aff45fa6724df208f188c33c7264a38aeb028b98","observation_id":"599c6f31-cc86-45d9-93bc-32d5d4cbf542","resolution":{"observed_at":"2026-08-10T15:40:41.169782Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:40:39.736586Z","title":"Mathstral","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.736586Z"},"links":{"citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:f65e45112c3f50c769440650404ec8d9c6cfcc6a44fcea451178d701fd0ea49c","observation_id":"a38806a0-1aaa-4941-8bc8-f89d5f7abca2","resolution":{"observed_at":"2026-08-10T15:40:39.736586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:40:41.137323Z","title":"Mistral-7b-instruct-v0.3","venue":null,"work_id":"ac936946-fd07-40fb-a4ad-f3854a9f5d0b","year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.741261Z"},"links":{"citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:a65a9ef5b78b4b71957b50247d626bfd5074ba7fb90bc9b7697f3609ec0581b4","observation_id":"2c317ae4-a445-4d8e-8806-c85647dfd5a3","resolution":{"observed_at":"2026-08-10T15:40:41.142404Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:40:41.119218Z","title":"Mistral large 2","venue":null,"work_id":"da5a3d9d-bf2e-4eb9-a411-e92f104460c0","year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.747037Z"},"links":{"citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:65dba68fc94077956d3366d01431671b0b403fec4b67577cf028187bd4e2b32b","observation_id":"3b1b4e39-5e71-4bf9-94e3-4a461cc34d4d","resolution":{"observed_at":"2026-08-10T15:40:41.124881Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:40:41.102051Z","title":"The future of ai: Trends and predictions","venue":null,"work_id":"56e18822-ebdf-48ab-b8cb-5728c0644a07","year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.751486Z"},"links":{"citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:050f34ff200191048b394ac91c42e36808107d7c62c21c29505413d15e23b9b0","observation_id":"aa9bbae8-0374-4506-9469-32bbc1d62eff","resolution":{"observed_at":"2026-08-10T15:40:41.107384Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:40:41.085767Z","title":"mistralai/mistral-small-instruct-2409","venue":null,"work_id":"9a1b9609-6e32-4b6a-b7de-08b6d219c2a2","year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.756854Z"},"links":{"citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:9249164e34a36b344fca4b47835f6f95cf61744cad52a0a7f73cdcc98e219c99","observation_id":"459c3e02-77f1-413d-9cdb-203c0485619b","resolution":{"observed_at":"2026-08-10T15:40:41.090764Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-10T15:40:39.761299Z","title":"Gpt-4 technical report","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.761299Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:a6436a113bc00bf4f30e8a09e7cf59bda3196429c169d1c8c4cc489825865160","observation_id":"d97bf66b-9ffc-40ba-b4a9-771b8196c7ff","resolution":{"observed_at":"2026-08-10T15:40:39.761299Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:40:39.765864Z","title":"Hello gpt-4o","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.765864Z"},"links":{"citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:b96c19c43844a91cc0d455339d2733d7d1253db3e8b968822fc56d402bd05953","observation_id":"457d8c92-56a7-459c-9e29-df681c5c9e25","resolution":{"observed_at":"2026-08-10T15:40:39.765864Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:40:39.770409Z","title":"Learning to reason with llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.770409Z"},"links":{"citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:00d5f0471333d6e1a2e2d9f8545a9ac60273ebace34e384747a021dfddeb6b04","observation_id":"21421130-e9a3-4aa4-8fe7-5d00c6be18b3","resolution":{"observed_at":"2026-08-10T15:40:39.770409Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:40:39.775015Z","title":"Training language models to follow instructions with human feedback","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.775015Z"},"links":{"citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:bd413a184f09a0bb947585f61d4da84555c57d81f578c5ff9da58d643fec0305","observation_id":"f11b28a2-617b-4ab1-af5e-66097d738444","resolution":{"observed_at":"2026-08-10T15:40:39.775015Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2103.07191","last_updated":"2021-04-15T06:11:12Z","snapshot_observed_at":"2026-08-07T13:11:17.300265Z","submitted_at":"2021-03-12T10:23:47Z","title":"Are NLP Models really able to Solve Simple Math Word Problems?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2103.07191","snapshot_observed_at":"2026-08-10T15:40:39.780365Z","title":"Are nlp models really able to solve simple math word problems? arXiv preprint arXiv:2103.07191, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.780365Z"},"links":{"cited_paper":"/paper/2103.07191","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:6d5b716ca4336e0bb1ee2e562c604e03107269a0c7b48727162eb5ab402fec49","observation_id":"024796c3-3452-4db2-ad50-c22106483284","resolution":{"observed_at":"2026-08-10T15:40:39.780365Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17681","last_updated":"2024-06-26T15:21:49Z","snapshot_observed_at":"2026-08-12T23:35:08.714368Z","submitted_at":"2024-06-25T16:13:53Z","title":"VarBench: Robust Language Model Benchmarking Through Dynamic Variable Perturbation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.17681","snapshot_observed_at":"2026-08-10T15:40:39.786212Z","title":"Varbench: Robust language model benchmarking through dynamic variable perturbation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.786212Z"},"links":{"cited_paper":"/paper/2406.17681","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:faae639fd4a96294ac262c5b7a4dcc4535144d2c59268d636bf54a7b71e68891","observation_id":"68217314-83f4-4253-a44b-77855e4a987a","resolution":{"observed_at":"2026-08-10T15:40:39.786212Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2202.07206","last_updated":"2022-05-24T02:08:16Z","snapshot_observed_at":"2026-08-14T06:40:07.804993Z","submitted_at":"2022-02-15T05:43:54Z","title":"Impact of Pretraining Term Frequencies on Few-Shot Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2202.07206","snapshot_observed_at":"2026-08-10T15:40:39.790994Z","title":"Impact of pretraining term frequencies on few-shot reasoning","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.790994Z"},"links":{"cited_paper":"/paper/2202.07206","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:20342d3fbd6cede134f1f09a61c1b0a0bd6247cae1a29167f1436efac4e3951f","observation_id":"a0d1a685-4a18-4770-9069-0b01d45904ae","resolution":{"observed_at":"2026-08-10T15:40:39.790994Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:40:41.037633Z","title":"To the cutoff","venue":null,"work_id":"6b29930b-7c21-46b7-a4fd-2807751a6aa8","year":2023},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.795631Z"},"links":{"citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:2512b12caafabce3268a8c092e241a3bd2b84ad5b5e22202238e4ca56e28b8fa","observation_id":"cce3f2c0-f7d3-4efc-bf83-9852a15f3771","resolution":{"observed_at":"2026-08-10T15:40:41.042710Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-10T15:40:39.800241Z","title":"Deepseekmath: Pushing the limits of mathematical reasoning in open language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.800241Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:49b771da40c76e901be90454c4af0dfdfbcbf110d9c7b9e54d9fb5a8f29958d3","observation_id":"1a048196-af5d-4ebc-a63b-cc9e3787ae95","resolution":{"observed_at":"2026-08-10T15:40:39.800241Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:40:39.805110Z","title":"Large language models can be easily distracted by irrelevant context","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.805110Z"},"links":{"citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:8ca70dc961c252735e84a7b0bdcbd504aea5b6a2b4f0bee5767172a6cb14c73d","observation_id":"cf56f8a0-24e0-4295-98c7-21a16c8832ae","resolution":{"observed_at":"2026-08-10T15:40:39.805110Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.19450","last_updated":"2024-02-29T18:48:18Z","snapshot_observed_at":"2026-08-13T04:06:49.278627Z","submitted_at":"2024-02-29T18:48:18Z","title":"Functional Benchmarks for Robust Evaluation of Reasoning Performance, and the Reasoning Gap","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.19450","snapshot_observed_at":"2026-08-10T15:40:39.809872Z","title":"Functional benchmarks for robust evaluation of reasoning performance, and the reasoning gap","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.809872Z"},"links":{"cited_paper":"/paper/2402.19450","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:cf460468ba36d1ab998b85a576b752266415fb4e52478bce653403a31117f67a","observation_id":"a7e5437e-8496-4c30-8a34-ceca47b43642","resolution":{"observed_at":"2026-08-10T15:40:39.809872Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.02884","last_updated":"2024-03-05T11:42:59Z","snapshot_observed_at":"2026-08-14T08:08:59.449816Z","submitted_at":"2024-03-05T11:42:59Z","title":"MathScale: Scaling Instruction Tuning for Mathematical Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.02884","snapshot_observed_at":"2026-08-10T15:40:39.814540Z","title":"Mathscale: Scaling instruction tuning for mathematical reasoning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.814540Z"},"links":{"cited_paper":"/paper/2403.02884","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:3c9fb5e740f16c31fa97c61c82ffa34bcaa82b72982f76335f05a2dd959aa972","observation_id":"ad55903f-86bb-4535-becd-65104a268541","resolution":{"observed_at":"2026-08-10T15:40:39.814540Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-10T15:40:39.820487Z","title":"Gemini: a family of highly capable multimodal models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.820487Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:34025033160fe383b18e81b692f77af22465fcb37228d0025ddfec1116a3ed42","observation_id":"934e076b-8366-41db-af36-00a957359448","resolution":{"observed_at":"2026-08-10T15:40:39.820487Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.13690","last_updated":"2024-12-23T17:32:21Z","snapshot_observed_at":"2026-08-13T00:20:11.438026Z","submitted_at":"2024-06-18T07:14:02Z","title":"DART-Math: Difficulty-Aware Rejection Tuning for Mathematical Problem-Solving","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.13690","snapshot_observed_at":"2026-08-10T15:40:39.825147Z","title":"Dart-math: Difficulty-aware rejection tuning for mathematical problem-solving","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.825147Z"},"links":{"cited_paper":"/paper/2407.13690","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:ac9c0157b70372ce5d9cde4641add360f266285825fc44289b9d2761146479bd","observation_id":"f2f0a032-c36b-4daa-8248-92a6c1db513f","resolution":{"observed_at":"2026-08-10T15:40:39.825147Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10635","last_updated":"2024-06-28T08:24:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-20T07:01:57Z","title":"SciBench: Evaluating College-Level Scientific Problem-Solving Abilities of Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.10635","snapshot_observed_at":"2026-08-10T15:40:39.830578Z","title":"Scibench: Evaluating college-level scientific problem-solving abilities of large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.830578Z"},"links":{"cited_paper":"/paper/2307.10635","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:a2d321702bdc701d7df6e8b4f77848f09e84a700e9084ebde2766964c0491f5b","observation_id":"2a1c0d03-b390-4700-83c0-5a4ed1f5084e","resolution":{"observed_at":"2026-08-10T15:40:39.830578Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.11171","last_updated":"2023-03-07T17:57:37Z","snapshot_observed_at":"2026-07-06T12:50:22.773056Z","submitted_at":"2022-03-21T17:48:52Z","title":"Self-Consistency Improves Chain of Thought Reasoning in Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.11171","snapshot_observed_at":"2026-08-10T15:40:39.835782Z","title":"Self-consistency improves chain of thought reasoning in language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.835782Z"},"links":{"cited_paper":"/paper/2203.11171","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:b81b8161897fca836ffc7b5749ca86960b4e190372d70f8bdf7d1eb95680ef20","observation_id":"89e4f12c-0e1e-45fe-a848-9dfbc2b60afe","resolution":{"observed_at":"2026-08-10T15:40:39.835782Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:40:41.010134Z","title":"Deep neural solver for math word problems","venue":null,"work_id":"d892cf2f-0250-4918-8df6-2ec19c476ca4","year":2017},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.840532Z"},"links":{"citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:b3f12fd0c6c7f943f51d08600fc86ed8cf0bed34439dde8f053c0038c191b6a1","observation_id":"1a947438-dea3-4669-ac75-a59931a82729","resolution":{"observed_at":"2026-08-10T15:40:41.015819Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-10T15:40:39.845175Z","title":"Mmlu-pro: A more robust and challenging multi-task language understanding benchmark","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.845175Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:7d96c9f03c91ba0e8b6ad41376870082c3afec3c21bb655bbc110e4c66d5e787","observation_id":"c01740dd-8ba5-46d9-bd61-ac4061053c35","resolution":{"observed_at":"2026-08-10T15:40:39.845175Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:40:39.850504Z","title":"Chain-of-thought prompting elicits reasoning in large language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.850504Z"},"links":{"citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:3767a460d6b806c470e5908d6f9cffdbf47958dbf8d3b9426510f91abf58a86e","observation_id":"2cb89d01-d296-4229-9d77-6f07b88bb2ab","resolution":{"observed_at":"2026-08-10T15:40:39.850504Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06209","last_updated":"2017-07-19T17:28:46Z","snapshot_observed_at":"2026-08-06T06:05:27.671303Z","submitted_at":"2017-07-19T17:28:46Z","title":"Crowdsourcing Multiple Choice Science Questions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1707.06209","snapshot_observed_at":"2026-08-10T15:40:39.855073Z","title":"Crowdsourcing multiple choice science questions","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.855073Z"},"links":{"cited_paper":"/paper/1707.06209","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:951d15f6ac2dd9ca94a008ec921f5dc2ff0a5fcda329027f830a76ae8a3d521a","observation_id":"ec7cec21-656e-4d3c-bdc1-f3db0a5496fd","resolution":{"observed_at":"2026-08-10T15:40:39.855073Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.19314","last_updated":"2025-04-18T19:36:00Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-27T16:47:42Z","title":"LiveBench: A Challenging, Contamination-Limited LLM Benchmark","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.19314","snapshot_observed_at":"2026-08-10T15:40:39.860944Z","title":"Livebench: A challenging, contamination-free llm benchmark","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.860944Z"},"links":{"cited_paper":"/paper/2406.19314","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:26d3b6e16d5ac489a90516425dec8289691cb23c064962c0d785d0c05d718eb6","observation_id":"b7273789-96d0-423a-b457-a92f31e57fcf","resolution":{"observed_at":"2026-08-10T15:40:39.860944Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.10528","last_updated":"2025-05-19T03:59:29Z","snapshot_observed_at":"2026-08-13T04:17:45.814493Z","submitted_at":"2024-02-16T09:29:50Z","title":"Can We Verify Step by Step for Incorrect Answer Detection?","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.10528","snapshot_observed_at":"2026-08-10T15:40:39.866556Z","title":"Can we verify step by step for incorrect answer detection? arXiv preprint arXiv:2402.10528, 2024 a","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.866556Z"},"links":{"cited_paper":"/paper/2402.10528","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:b4bf1cf1cf4d11bb02649f138b18d9b91f5ca316923b0c4097977cbf74ef72c7","observation_id":"9cd00303-8f6a-4d48-b38c-ded731bf435a","resolution":{"observed_at":"2026-08-10T15:40:39.866556Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.14804","last_updated":"2025-02-26T02:21:40Z","snapshot_observed_at":"2026-08-14T11:31:25.285986Z","submitted_at":"2024-05-23T17:13:50Z","title":"Can LLMs Solve longer Math Word Problems Better?","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.14804","snapshot_observed_at":"2026-08-10T15:40:39.872020Z","title":"Can llms solve longer math word problems better? arXiv preprint arXiv:2405.14804, 2024 b","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.872020Z"},"links":{"cited_paper":"/paper/2405.14804","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:e64408fab6d92b1b50f8f4a343de0c2f53816e4978f773c3a3c903327054abe6","observation_id":"23f0feda-a3e9-42b3-95f5-ecf2217b8a83","resolution":{"observed_at":"2026-08-10T15:40:39.872020Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01524","last_updated":"2025-06-28T04:34:18Z","snapshot_observed_at":"2026-08-12T22:52:44.774063Z","submitted_at":"2024-09-03T01:40:21Z","title":"S^3cMath: Spontaneous Step-level Self-correction Makes Large Language Models Better Mathematical Reasoners","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01524","snapshot_observed_at":"2026-08-10T15:40:39.876942Z","title":"Spontaneous step-level self-correction makes large language models better mathematical reasoners","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.876942Z"},"links":{"cited_paper":"/paper/2409.01524","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:433d49f85ab61ef80cff8b51919619d9542edb46eda430a75dfe4f53ed85a007","observation_id":"2220d358-a814-4ef2-94e9-06802039777d","resolution":{"observed_at":"2026-08-10T15:40:39.876942Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10671","last_updated":"2024-09-10T13:25:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-07-15T12:35:42Z","title":"Qwen2 Technical Report","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.10671","snapshot_observed_at":"2026-08-10T15:40:39.882278Z","title":"Qwen2 technical report","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.882278Z"},"links":{"cited_paper":"/paper/2407.10671","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:5344af50cda80a61801fb23b607d92cf768779daf398a50f01ee643504212208","observation_id":"026e68af-1ce4-442b-8fda-b46bd6829f18","resolution":{"observed_at":"2026-08-10T15:40:39.882278Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12122","last_updated":"2024-09-18T16:45:37Z","snapshot_observed_at":"2026-08-13T17:31:32.138475Z","submitted_at":"2024-09-18T16:45:37Z","title":"Qwen2.5-Math Technical Report: Toward Mathematical Expert Model via Self-Improvement","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12122","snapshot_observed_at":"2026-08-10T15:40:39.887051Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.887051Z"},"links":{"cited_paper":"/paper/2409.12122","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:6a9a6292e7ef19ae5935f45e0ec31afe4dd1d9d1d81caf0674df330e36baffe9","observation_id":"47b04974-d71d-41ad-a51f-c9a1a33f51dd","resolution":{"observed_at":"2026-08-10T15:40:39.887051Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.12284","last_updated":"2024-05-03T17:36:07Z","snapshot_observed_at":"2026-08-13T10:57:13.012119Z","submitted_at":"2023-09-21T17:45:42Z","title":"MetaMath: Bootstrap Your Own Mathematical Questions for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.12284","snapshot_observed_at":"2026-08-10T15:40:39.891943Z","title":"Metamath: Bootstrap your own mathematical questions for large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.891943Z"},"links":{"cited_paper":"/paper/2309.12284","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:434ed463cb08f5c4dae2c81c1d519420adf318be975ba7eb9751acedf93fa40d","observation_id":"4e1cef28-7bac-4abb-a63d-fa80d8a41e69","resolution":{"observed_at":"2026-08-10T15:40:39.891943Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.05653","last_updated":"2023-10-03T02:48:42Z","snapshot_observed_at":"2026-08-07T14:43:09.380195Z","submitted_at":"2023-09-11T17:47:22Z","title":"MAmmoTH: Building Math Generalist Models through Hybrid Instruction Tuning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.05653","snapshot_observed_at":"2026-08-10T15:40:39.897050Z","title":"Mammoth: Building math generalist models through hybrid instruction tuning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.897050Z"},"links":{"cited_paper":"/paper/2309.05653","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:2ae3edcf616f2fa925165f3a1f8c05a998aab25cbebbd6050a402a3425dd66a6","observation_id":"b89a4fc6-2a52-4f64-830b-90fc643e0569","resolution":{"observed_at":"2026-08-10T15:40:39.897050Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:40:40.980592Z","title":"Mmmu: A massive multi-discipline multimodal understanding and reasoning benchmark for expert agi","venue":null,"work_id":"93fdfcbf-2e51-4b9d-95fa-b331b3fac33a","year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.901836Z"},"links":{"citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:dd222698176c52ce56f43d2e68594c1de407295998cd96f4561efa8e70a26cd5","observation_id":"a245528d-1133-4eff-8ad8-e3739da4b8f6","resolution":{"observed_at":"2026-08-10T15:40:40.987548Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.02813","last_updated":"2025-05-22T08:22:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-04T15:31:26Z","title":"MMMU-Pro: A More Robust Multi-discipline Multimodal Understanding Benchmark","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.02813","snapshot_observed_at":"2026-08-10T15:40:39.907302Z","title":"Mmmu-pro: A more robust multi-discipline multimodal understanding benchmark","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.907302Z"},"links":{"cited_paper":"/paper/2409.02813","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:30aabd07fa7f2541f8a16d881aa9f59e6a4884bd0638db24db1fd5e4092f70f4","observation_id":"927131eb-c1e4-4242-935b-097e1159a221","resolution":{"observed_at":"2026-08-10T15:40:39.907302Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00332","last_updated":"2024-11-22T22:27:49Z","snapshot_observed_at":"2026-08-14T07:41:49.296836Z","submitted_at":"2024-05-01T05:52:05Z","title":"A Careful Examination of Large Language Model Performance on Grade School Arithmetic","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00332","snapshot_observed_at":"2026-08-10T15:40:39.912031Z","title":"A careful examination of large language model performance on grade school arithmetic","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.912031Z"},"links":{"cited_paper":"/paper/2405.00332","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:618ce87c14018731417859e12c93936870d1f79e47837c4f84d455225a52b48f","observation_id":"29d7a177-961d-447e-b8a0-95c4f6d97ac3","resolution":{"observed_at":"2026-08-10T15:40:39.912031Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.04371","last_updated":"2026-05-21T08:45:38Z","snapshot_observed_at":"2026-08-14T11:14:00.151653Z","submitted_at":"2023-08-08T16:18:20Z","title":"Cumulative Reasoning with Large Language Models","version":11},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.04371","snapshot_observed_at":"2026-08-10T15:40:39.916871Z","title":"Cumulative reasoning with large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":75,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.916871Z"},"links":{"cited_paper":"/paper/2308.04371","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:3c187f8e94a98720eb7c03594566b4bebbae5076c7526920c97028c0019bc52c","observation_id":"05349184-bc8f-46eb-ab58-95304d74812e","resolution":{"observed_at":"2026-08-10T15:40:39.916871Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09797","last_updated":"2024-10-07T04:28:04Z","snapshot_observed_at":"2026-08-13T11:59:18.236451Z","submitted_at":"2023-04-19T16:29:48Z","title":"Progressive-Hint Prompting Improves Reasoning in Large Language Models","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09797","snapshot_observed_at":"2026-08-10T15:40:39.921948Z","title":"Progressive-hint prompting improves reasoning in large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":76,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.921948Z"},"links":{"cited_paper":"/paper/2304.09797","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:fbc3c7b1d4099b3d5c01b23fba1cec5cd67011dc7f692be32969b44ee9ecd136","observation_id":"6c3ec3ec-f72d-4321-bbf7-c7c786e381cb","resolution":{"observed_at":"2026-08-10T15:40:39.921948Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.07921","last_updated":"2023-08-15T17:58:45Z","snapshot_observed_at":"2026-08-13T10:34:06.970583Z","submitted_at":"2023-08-15T17:58:45Z","title":"Solving Challenging Math Word Problems Using GPT-4 Code Interpreter with Code-based Self-Verification","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.07921","snapshot_observed_at":"2026-08-10T15:40:39.926795Z","title":"Solving challenging math word problems using gpt-4 code interpreter with code-based self-verification","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.926795Z"},"links":{"cited_paper":"/paper/2308.07921","citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:8f5cb2cd7fa3ab200fd9d62ec0e2ed40c9263a0dd0205b3f01cdf34008478ee7","observation_id":"23df265d-f769-483c-a362-c3dfa64b58b4","resolution":{"observed_at":"2026-08-10T15:40:39.926795Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:40:39.931716Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":78,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.931716Z"},"links":{"citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:c363f53878d17534de902d8185451db32990058472edb56c814c0944f16f0ea5","observation_id":"ef799b5d-3e13-4643-8f81-5b823d7740c9","resolution":{"observed_at":"2026-08-10T15:40:39.931716Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:40:39.937620Z","title":"@esa (Ref","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":79,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.937620Z"},"links":{"citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:e19e96f1d5fa2c3b94061f44fea23cc3503c1fae422661702c99ad14d6bfcceb","observation_id":"00c936af-350d-401d-9171-b91ecdd63caf","resolution":{"observed_at":"2026-08-10T15:40:39.937620Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:40:39.944207Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":80,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.944207Z"},"links":{"citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:12a8cce3c84ca45af4cf42c89329714365bfe9f6afac5ec4c5fed9391717d7e8","observation_id":"fb76f3c2-b2cc-4f7d-afa2-9552682e12ae","resolution":{"observed_at":"2026-08-10T15:40:39.944207Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T15:40:39.950474Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models","version":2},"reference_index":81,"source":"arxiv_source","source_observed_at":"2026-08-10T15:40:39.950474Z"},"links":{"citing_paper":"/paper/2501.13766"},"observation_digest":"sha256:11712683af912242ebce1caa74baf88e45a357c94d06435fba8a25bd3ab4a56f","observation_id":"b46613b4-3af1-405e-95c1-bd144fb6d148","resolution":{"observed_at":"2026-08-10T15:40:39.950474Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2501.13766","last_updated":"2025-02-25T08:15:43Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-13T20:24:00.439807Z","submitted_at":"2025-01-23T15:46:43Z","title":"UGMathBench: A Diverse and Dynamic Benchmark for Undergraduate-Level Mathematical Reasoning with Large Language Models"},"reference_resolution":{"displayed":81,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":68,"verified_exact":0,"verified_fuzzy":13},"total_outbound_references":81},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"thesis":"As of 14 August 2026, this Paper Citation Record lists 81 of 81 outbound references and 4 inbound Pith citation observations for arXiv:2501.13766."}