{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:EDJVQGSRJUGAYYUA74POGAXRB7","short_pith_number":"pith:EDJVQGSR","schema_version":"1.0","canonical_sha256":"20d3581a514d0c0c6280ff1ee302f10fdd6c20b5d63780647ec65155349f9c16","source":{"kind":"arxiv","id":"2502.09054","version":2},"attestation_state":"computed","paper":{"title":"Cost-Saving LLM Cascades with Early Abstention","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Matt Thomson, Michael J. Zellinger, Rex Liu","submitted_at":"2025-02-13T08:08:39Z","abstract_excerpt":"LLM cascades deploy small LLMs to answer most queries, limiting the use of large and expensive LLMs to difficult queries. This approach can significantly reduce costs without impacting performance. However, risk-sensitive domains such as finance or medicine place an additional premium on avoiding model errors. Since even the most expensive models are susceptible to making mistakes, applications in these domains benefit from allowing LLM systems to completely abstain from answering difficult queries. Introducing abstention poses a design question for LLM cascades: should abstention only be allo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.09054","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-02-13T08:08:39Z","cross_cats_sorted":[],"title_canon_sha256":"474bfc97df44ee0c4ff91d8ce8f57352410c557880ab3267a32a1794c2808b14","abstract_canon_sha256":"474afd3a79e679b5288225b0db92c4d5b53b9f6b293e171544674a9fdc09f4db"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:41:20.566460Z","signature_b64":"WMPYrakQlyddhrToajI5fwIVIvDHkPWmcaavXs+9g1YheLH9oOR+SmdGI/DPt4+F5yLE+V6DMbHpefripCVTDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"20d3581a514d0c0c6280ff1ee302f10fdd6c20b5d63780647ec65155349f9c16","last_reissued_at":"2026-07-05T10:41:20.565933Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:41:20.565933Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Cost-Saving LLM Cascades with Early Abstention","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Matt Thomson, Michael J. Zellinger, Rex Liu","submitted_at":"2025-02-13T08:08:39Z","abstract_excerpt":"LLM cascades deploy small LLMs to answer most queries, limiting the use of large and expensive LLMs to difficult queries. This approach can significantly reduce costs without impacting performance. However, risk-sensitive domains such as finance or medicine place an additional premium on avoiding model errors. Since even the most expensive models are susceptible to making mistakes, applications in these domains benefit from allowing LLM systems to completely abstain from answering difficult queries. Introducing abstention poses a design question for LLM cascades: should abstention only be allo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.09054","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.09054/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.09054","created_at":"2026-07-05T10:41:20.565989+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.09054v2","created_at":"2026-07-05T10:41:20.565989+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.09054","created_at":"2026-07-05T10:41:20.565989+00:00"},{"alias_kind":"pith_short_12","alias_value":"EDJVQGSRJUGA","created_at":"2026-07-05T10:41:20.565989+00:00"},{"alias_kind":"pith_short_16","alias_value":"EDJVQGSRJUGAYYUA","created_at":"2026-07-05T10:41:20.565989+00:00"},{"alias_kind":"pith_short_8","alias_value":"EDJVQGSR","created_at":"2026-07-05T10:41:20.565989+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06503","citing_title":"Doomed from the Start: Early Abort of LLM Agent Episodes via a Recall-Controlled Probe Cascade","ref_index":25,"is_internal_anchor":true},{"citing_arxiv_id":"2606.12587","citing_title":"Strategic Decision Support for AI Agents","ref_index":96,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18796","citing_title":"UCCI: Calibrated Uncertainty for Cost-Optimal LLM Cascade Routing","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15206","citing_title":"AgentStop: Terminating Local AI Agents Early to Save Energy in Consumer Devices","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03216","citing_title":"BAS: A Decision-Theoretic Approach to Evaluating Large Language Model Confidence","ref_index":55,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EDJVQGSRJUGAYYUA74POGAXRB7","json":"https://pith.science/pith/EDJVQGSRJUGAYYUA74POGAXRB7.json","graph_json":"https://pith.science/api/pith-number/EDJVQGSRJUGAYYUA74POGAXRB7/graph.json","events_json":"https://pith.science/api/pith-number/EDJVQGSRJUGAYYUA74POGAXRB7/events.json","paper":"https://pith.science/paper/EDJVQGSR"},"agent_actions":{"view_html":"https://pith.science/pith/EDJVQGSRJUGAYYUA74POGAXRB7","download_json":"https://pith.science/pith/EDJVQGSRJUGAYYUA74POGAXRB7.json","view_paper":"https://pith.science/paper/EDJVQGSR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.09054&json=true","fetch_graph":"https://pith.science/api/pith-number/EDJVQGSRJUGAYYUA74POGAXRB7/graph.json","fetch_events":"https://pith.science/api/pith-number/EDJVQGSRJUGAYYUA74POGAXRB7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EDJVQGSRJUGAYYUA74POGAXRB7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EDJVQGSRJUGAYYUA74POGAXRB7/action/storage_attestation","attest_author":"https://pith.science/pith/EDJVQGSRJUGAYYUA74POGAXRB7/action/author_attestation","sign_citation":"https://pith.science/pith/EDJVQGSRJUGAYYUA74POGAXRB7/action/citation_signature","submit_replication":"https://pith.science/pith/EDJVQGSRJUGAYYUA74POGAXRB7/action/replication_record"}},"created_at":"2026-07-05T10:41:20.565989+00:00","updated_at":"2026-07-05T10:41:20.565989+00:00"}