{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KIERGE5ZX5PZVQKYUTBMGPUAN7","short_pith_number":"pith:KIERGE5Z","schema_version":"1.0","canonical_sha256":"52091313b9bf5f9ac158a4c2c33e806fd39cfbcb6ba84936ecdf38a898af6f6e","source":{"kind":"arxiv","id":"2412.18547","version":5},"attestation_state":"computed","paper":{"title":"Token-Budget-Aware LLM Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chunrong Fang, Shiqing Ma, Shiyu Zhao, Tingxu Han, Zhenting Wang, Zhenyu Chen","submitted_at":"2024-12-24T16:55:45Z","abstract_excerpt":"Reasoning is critical for large language models (LLMs) to excel in a wide range of tasks. While methods like Chain-of-Thought (CoT) reasoning and enhance LLM performance by decomposing problems into intermediate steps, they also incur significant overhead in token usage, leading to increased costs. We find that the reasoning process of current LLMs is unnecessarily lengthy and it can be compressed by including a reasonable token budget in the prompt, but the choice of token budget plays a crucial role in the actual compression effectiveness. We then propose a token-budget-aware LLM reasoning f"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.18547","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-12-24T16:55:45Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"79aab1c0b6849086765f2bf7efadd76031282fd04de325c5e2c1068d8ce443cc","abstract_canon_sha256":"22097ab5bef7ac1821a1c158ca0468ceb194ddd181fcf56af4017a8535f7fe3e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:14:01.301698Z","signature_b64":"Wf4Yo+Jj3wTqqKA1MfTuNN4eP15ZQPoF6Y7Yx+pAynfZ/TTiAq8zRHiH0OkbtcYdUgI1LFlY66xKrQqEEY1dCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"52091313b9bf5f9ac158a4c2c33e806fd39cfbcb6ba84936ecdf38a898af6f6e","last_reissued_at":"2026-07-05T11:14:01.301075Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:14:01.301075Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Token-Budget-Aware LLM Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chunrong Fang, Shiqing Ma, Shiyu Zhao, Tingxu Han, Zhenting Wang, Zhenyu Chen","submitted_at":"2024-12-24T16:55:45Z","abstract_excerpt":"Reasoning is critical for large language models (LLMs) to excel in a wide range of tasks. While methods like Chain-of-Thought (CoT) reasoning and enhance LLM performance by decomposing problems into intermediate steps, they also incur significant overhead in token usage, leading to increased costs. We find that the reasoning process of current LLMs is unnecessarily lengthy and it can be compressed by including a reasonable token budget in the prompt, but the choice of token budget plays a crucial role in the actual compression effectiveness. We then propose a token-budget-aware LLM reasoning f"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.18547","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.18547/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.18547","created_at":"2026-07-05T11:14:01.301145+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.18547v5","created_at":"2026-07-05T11:14:01.301145+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.18547","created_at":"2026-07-05T11:14:01.301145+00:00"},{"alias_kind":"pith_short_12","alias_value":"KIERGE5ZX5PZ","created_at":"2026-07-05T11:14:01.301145+00:00"},{"alias_kind":"pith_short_16","alias_value":"KIERGE5ZX5PZVQKY","created_at":"2026-07-05T11:14:01.301145+00:00"},{"alias_kind":"pith_short_8","alias_value":"KIERGE5Z","created_at":"2026-07-05T11:14:01.301145+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":17,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06906","citing_title":"The Harness Effect: How Orchestration Design Sets the Token Economics of Enterprise Agentic AI","ref_index":12,"is_internal_anchor":true},{"citing_arxiv_id":"2605.23926","citing_title":"How Much Thinking is Enough? Quantifying and Understanding Redundancy in LLM Reasoning","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20295","citing_title":"Token-Operations-Oriented Inference Optimization Techniques for Large Models","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03800","citing_title":"Trading Human Curation for Synthetic Augmentation in RLVR","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02282","citing_title":"POIROT: Interrogating Agents for Failure Detection in Multi-Agent Systems","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11470","citing_title":"The Periodic Table of LLM Reasoning: A Structured Survey of Reasoning Paradigms, Methods, and Failure Modes","ref_index":87,"is_internal_anchor":false},{"citing_arxiv_id":"2505.15858","citing_title":"Search-Based Multi-Trajectory Refinement for Safe C-to-Rust Translation with Large Language Models","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2509.14004","citing_title":"Early Stopping Chain-of-thoughts in Large Language Models","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2509.05489","citing_title":"Self-Aligned Reward: Towards Effective and Efficient Reasoners","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2509.21743","citing_title":"Retrieval-of-Thought: Efficient Reasoning via Reusing Thoughts","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2503.16419","citing_title":"Stop Overthinking: A Survey on Efficient Reasoning for Large Language Models","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03679","citing_title":"LightThinker++: From Reasoning Compression to Memory Management","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2412.21187","citing_title":"Do NOT Think That Much for 2+3=? On the Overthinking of o1-Like LLMs","ref_index":279,"is_internal_anchor":false},{"citing_arxiv_id":"2503.09567","citing_title":"Towards Reasoning Era: A Survey of Long Chain-of-Thought for Reasoning Large Language Models","ref_index":236,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06165","citing_title":"Post Reasoning: Improving the Performance of Non-Thinking Models at No Cost","ref_index":107,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21764","citing_title":"Thinking with Reasoning Skills: Fewer Tokens, More Accuracy","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08290","citing_title":"Tokalator: A Context Engineering Toolkit for Artificial Intelligence Coding Assistants","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KIERGE5ZX5PZVQKYUTBMGPUAN7","json":"https://pith.science/pith/KIERGE5ZX5PZVQKYUTBMGPUAN7.json","graph_json":"https://pith.science/api/pith-number/KIERGE5ZX5PZVQKYUTBMGPUAN7/graph.json","events_json":"https://pith.science/api/pith-number/KIERGE5ZX5PZVQKYUTBMGPUAN7/events.json","paper":"https://pith.science/paper/KIERGE5Z"},"agent_actions":{"view_html":"https://pith.science/pith/KIERGE5ZX5PZVQKYUTBMGPUAN7","download_json":"https://pith.science/pith/KIERGE5ZX5PZVQKYUTBMGPUAN7.json","view_paper":"https://pith.science/paper/KIERGE5Z","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.18547&json=true","fetch_graph":"https://pith.science/api/pith-number/KIERGE5ZX5PZVQKYUTBMGPUAN7/graph.json","fetch_events":"https://pith.science/api/pith-number/KIERGE5ZX5PZVQKYUTBMGPUAN7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KIERGE5ZX5PZVQKYUTBMGPUAN7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KIERGE5ZX5PZVQKYUTBMGPUAN7/action/storage_attestation","attest_author":"https://pith.science/pith/KIERGE5ZX5PZVQKYUTBMGPUAN7/action/author_attestation","sign_citation":"https://pith.science/pith/KIERGE5ZX5PZVQKYUTBMGPUAN7/action/citation_signature","submit_replication":"https://pith.science/pith/KIERGE5ZX5PZVQKYUTBMGPUAN7/action/replication_record"}},"created_at":"2026-07-05T11:14:01.301145+00:00","updated_at":"2026-07-05T11:14:01.301145+00:00"}