{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:R4HL42AJK7Q4XDLSA3B27ENHD2","short_pith_number":"pith:R4HL42AJ","schema_version":"1.0","canonical_sha256":"8f0ebe680957e1cb8d7206c3af91a71e8d71eba21a4c75558d0c2f8e13032222","source":{"kind":"arxiv","id":"2405.16265","version":4},"attestation_state":"computed","paper":{"title":"MindStar: Enhancing Math Reasoning in Pre-trained LLMs at Inference Time","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Amirreza Kazemi, Boxing Chen, Dong Li, Feng Wen, Jianye Hao, Jikun Kang, Jun Yao, Qianyi Sun, Quan He, Xi Chen, Xin Zhe Li, Xu He","submitted_at":"2024-05-25T15:07:33Z","abstract_excerpt":"Although Large Language Models (LLMs) achieve remarkable performance across various tasks, they often struggle with complex reasoning tasks, such as answering mathematical questions. Recent efforts to address this issue have primarily focused on leveraging mathematical datasets through supervised fine-tuning or self-improvement techniques. However, these methods often depend on high-quality datasets that are difficult to prepare, or they require substantial computational resources for fine-tuning. Inspired by findings that LLMs know how to produce the right answer but struggle to select the co"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.16265","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-25T15:07:33Z","cross_cats_sorted":[],"title_canon_sha256":"9ba1768ebbb31a96b1bf79045f2305c2b8025256ecf81fce496c2835140cf834","abstract_canon_sha256":"dd4fafe37fbc6fb0cebb2d52f4085551aebc317da2c82d4246bc443094b63a16"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:37:06.971822Z","signature_b64":"vEaqcrQqO2aUaqjlTR5dbcA1ie8+kAjw+lcj3sT/1QxRYtElWrp1lA5Rv/7wtaJodP2FtxgfgaVUl/SJmkOTAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8f0ebe680957e1cb8d7206c3af91a71e8d71eba21a4c75558d0c2f8e13032222","last_reissued_at":"2026-07-05T08:37:06.971378Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:37:06.971378Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MindStar: Enhancing Math Reasoning in Pre-trained LLMs at Inference Time","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Amirreza Kazemi, Boxing Chen, Dong Li, Feng Wen, Jianye Hao, Jikun Kang, Jun Yao, Qianyi Sun, Quan He, Xi Chen, Xin Zhe Li, Xu He","submitted_at":"2024-05-25T15:07:33Z","abstract_excerpt":"Although Large Language Models (LLMs) achieve remarkable performance across various tasks, they often struggle with complex reasoning tasks, such as answering mathematical questions. Recent efforts to address this issue have primarily focused on leveraging mathematical datasets through supervised fine-tuning or self-improvement techniques. However, these methods often depend on high-quality datasets that are difficult to prepare, or they require substantial computational resources for fine-tuning. Inspired by findings that LLMs know how to produce the right answer but struggle to select the co"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.16265","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.16265/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.16265","created_at":"2026-07-05T08:37:06.971433+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.16265v4","created_at":"2026-07-05T08:37:06.971433+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.16265","created_at":"2026-07-05T08:37:06.971433+00:00"},{"alias_kind":"pith_short_12","alias_value":"R4HL42AJK7Q4","created_at":"2026-07-05T08:37:06.971433+00:00"},{"alias_kind":"pith_short_16","alias_value":"R4HL42AJK7Q4XDLS","created_at":"2026-07-05T08:37:06.971433+00:00"},{"alias_kind":"pith_short_8","alias_value":"R4HL42AJ","created_at":"2026-07-05T08:37:06.971433+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.24989","citing_title":"Selective Test-Time Compute Scaling for Click-Through Rate Prediction via Uncertainty-Triggered Feature Path Exploration","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16302","citing_title":"Reducing Credit Assignment Variance via Counterfactual Reasoning Paths","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2406.06592","citing_title":"Improve Mathematical Reasoning in Language Models by Automated Process Supervision","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2502.17419","citing_title":"From System 1 to System 2: A Survey of Reasoning Large Language Models","ref_index":137,"is_internal_anchor":false},{"citing_arxiv_id":"2407.21787","citing_title":"Large Language Monkeys: Scaling Inference Compute with Repeated Sampling","ref_index":38,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/R4HL42AJK7Q4XDLSA3B27ENHD2","json":"https://pith.science/pith/R4HL42AJK7Q4XDLSA3B27ENHD2.json","graph_json":"https://pith.science/api/pith-number/R4HL42AJK7Q4XDLSA3B27ENHD2/graph.json","events_json":"https://pith.science/api/pith-number/R4HL42AJK7Q4XDLSA3B27ENHD2/events.json","paper":"https://pith.science/paper/R4HL42AJ"},"agent_actions":{"view_html":"https://pith.science/pith/R4HL42AJK7Q4XDLSA3B27ENHD2","download_json":"https://pith.science/pith/R4HL42AJK7Q4XDLSA3B27ENHD2.json","view_paper":"https://pith.science/paper/R4HL42AJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.16265&json=true","fetch_graph":"https://pith.science/api/pith-number/R4HL42AJK7Q4XDLSA3B27ENHD2/graph.json","fetch_events":"https://pith.science/api/pith-number/R4HL42AJK7Q4XDLSA3B27ENHD2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/R4HL42AJK7Q4XDLSA3B27ENHD2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/R4HL42AJK7Q4XDLSA3B27ENHD2/action/storage_attestation","attest_author":"https://pith.science/pith/R4HL42AJK7Q4XDLSA3B27ENHD2/action/author_attestation","sign_citation":"https://pith.science/pith/R4HL42AJK7Q4XDLSA3B27ENHD2/action/citation_signature","submit_replication":"https://pith.science/pith/R4HL42AJK7Q4XDLSA3B27ENHD2/action/replication_record"}},"created_at":"2026-07-05T08:37:06.971433+00:00","updated_at":"2026-07-05T08:37:06.971433+00:00"}