{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:25MHO4W2BXGRX3PZWMOOQNEZOE","short_pith_number":"pith:25MHO4W2","schema_version":"1.0","canonical_sha256":"d7587772da0dcd1bedf9b31ce83499710c9328eb90a2129e50b0dade2e9b85df","source":{"kind":"arxiv","id":"2312.11462","version":5},"attestation_state":"computed","paper":{"title":"Cascade Speculative Drafting for Even Faster LLM Inference","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Chenkai Sun, Jiacheng Lin, Jie Huang, Kevin Chen-Chuan Chang, Xiaocong Yang, Ziyi Chen","submitted_at":"2023-12-18T18:59:46Z","abstract_excerpt":"Introduced to enhance the efficiency of large language model (LLM) inference, speculative decoding operates by having a smaller model generate a draft. A larger target model then reviews this draft to align with its output, and any acceptance by the target model results in a reduction of the number of the target model runs, ultimately improving efficiency. However, the drafting process in speculative decoding includes slow autoregressive generation and allocates equal time to generating tokens, irrespective of their importance. These inefficiencies collectively contribute to the suboptimal per"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.11462","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-12-18T18:59:46Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"3cf8cafed022ed3de01ae579162c5c6261e26edfb72aa174decbf793f05c15db","abstract_canon_sha256":"40e2f8400945a9befd3cf73b84cf026ded04a4bf939808ad6b903074b3723f44"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:36:18.850116Z","signature_b64":"LtMFpUGvVALBGl7DXmkHMfwWkH7Td0wfiIJU4Hn18apTWvXJw5hl1fuBxavcUVZrsBrfTvNc3bBEBXq3LpW2CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d7587772da0dcd1bedf9b31ce83499710c9328eb90a2129e50b0dade2e9b85df","last_reissued_at":"2026-07-05T11:36:18.849629Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:36:18.849629Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Cascade Speculative Drafting for Even Faster LLM Inference","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Chenkai Sun, Jiacheng Lin, Jie Huang, Kevin Chen-Chuan Chang, Xiaocong Yang, Ziyi Chen","submitted_at":"2023-12-18T18:59:46Z","abstract_excerpt":"Introduced to enhance the efficiency of large language model (LLM) inference, speculative decoding operates by having a smaller model generate a draft. A larger target model then reviews this draft to align with its output, and any acceptance by the target model results in a reduction of the number of the target model runs, ultimately improving efficiency. However, the drafting process in speculative decoding includes slow autoregressive generation and allocates equal time to generating tokens, irrespective of their importance. These inefficiencies collectively contribute to the suboptimal per"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.11462","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.11462/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.11462","created_at":"2026-07-05T11:36:18.849695+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.11462v5","created_at":"2026-07-05T11:36:18.849695+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.11462","created_at":"2026-07-05T11:36:18.849695+00:00"},{"alias_kind":"pith_short_12","alias_value":"25MHO4W2BXGR","created_at":"2026-07-05T11:36:18.849695+00:00"},{"alias_kind":"pith_short_16","alias_value":"25MHO4W2BXGRX3PZ","created_at":"2026-07-05T11:36:18.849695+00:00"},{"alias_kind":"pith_short_8","alias_value":"25MHO4W2","created_at":"2026-07-05T11:36:18.849695+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.18796","citing_title":"UCCI: Calibrated Uncertainty for Cost-Optimal LLM Cascade Routing","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16874","citing_title":"Reasoning Can Be Restored by Correcting a Few Decision Tokens","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2404.14294","citing_title":"A Survey on Efficient Inference for Large Language Models","ref_index":245,"is_internal_anchor":false},{"citing_arxiv_id":"2401.15077","citing_title":"EAGLE: Speculative Sampling Requires Rethinking Feature Uncertainty","ref_index":49,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/25MHO4W2BXGRX3PZWMOOQNEZOE","json":"https://pith.science/pith/25MHO4W2BXGRX3PZWMOOQNEZOE.json","graph_json":"https://pith.science/api/pith-number/25MHO4W2BXGRX3PZWMOOQNEZOE/graph.json","events_json":"https://pith.science/api/pith-number/25MHO4W2BXGRX3PZWMOOQNEZOE/events.json","paper":"https://pith.science/paper/25MHO4W2"},"agent_actions":{"view_html":"https://pith.science/pith/25MHO4W2BXGRX3PZWMOOQNEZOE","download_json":"https://pith.science/pith/25MHO4W2BXGRX3PZWMOOQNEZOE.json","view_paper":"https://pith.science/paper/25MHO4W2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.11462&json=true","fetch_graph":"https://pith.science/api/pith-number/25MHO4W2BXGRX3PZWMOOQNEZOE/graph.json","fetch_events":"https://pith.science/api/pith-number/25MHO4W2BXGRX3PZWMOOQNEZOE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/25MHO4W2BXGRX3PZWMOOQNEZOE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/25MHO4W2BXGRX3PZWMOOQNEZOE/action/storage_attestation","attest_author":"https://pith.science/pith/25MHO4W2BXGRX3PZWMOOQNEZOE/action/author_attestation","sign_citation":"https://pith.science/pith/25MHO4W2BXGRX3PZWMOOQNEZOE/action/citation_signature","submit_replication":"https://pith.science/pith/25MHO4W2BXGRX3PZWMOOQNEZOE/action/replication_record"}},"created_at":"2026-07-05T11:36:18.849695+00:00","updated_at":"2026-07-05T11:36:18.849695+00:00"}