{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:4KFBPBKNHWJO4E46DZBDPZBVLX","short_pith_number":"pith:4KFBPBKN","schema_version":"1.0","canonical_sha256":"e28a17854d3d92ee139e1e4237e4355ddb336bd6e20c3e6bdaf3008a18385e09","source":{"kind":"arxiv","id":"2210.05528","version":1},"attestation_state":"computed","paper":{"title":"Model Cascading: Towards Jointly Improving Efficiency and Accuracy of NLP Systems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chitta Baral, Neeraj Varshney","submitted_at":"2022-10-11T15:17:52Z","abstract_excerpt":"Do all instances need inference through the big models for a correct prediction? Perhaps not; some instances are easy and can be answered correctly by even small capacity models. This provides opportunities for improving the computational efficiency of systems. In this work, we present an explorative study on 'model cascading', a simple technique that utilizes a collection of models of varying capacities to accurately yet efficiently output predictions. Through comprehensive experiments in multiple task settings that differ in the number of models available for cascading (K value), we show tha"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2210.05528","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2022-10-11T15:17:52Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"bfc4feedc2eb12307dfbd312ac6447c316d2190a35c0ceb69c9e70bf2888a747","abstract_canon_sha256":"f9e1888939bd058f544747e52a9babe88fac4d7930bfe6d145364a01e7438b99"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:05:17.680843Z","signature_b64":"1DlMzL39tsijOF+csblMHdHDlDLncVtPmbt8/AlkEpnC4tQS5tczsp/GS/scWbCnrXqUwjI3CtzRMkYQC3MtCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e28a17854d3d92ee139e1e4237e4355ddb336bd6e20c3e6bdaf3008a18385e09","last_reissued_at":"2026-07-05T05:05:17.680354Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:05:17.680354Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Model Cascading: Towards Jointly Improving Efficiency and Accuracy of NLP Systems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chitta Baral, Neeraj Varshney","submitted_at":"2022-10-11T15:17:52Z","abstract_excerpt":"Do all instances need inference through the big models for a correct prediction? Perhaps not; some instances are easy and can be answered correctly by even small capacity models. This provides opportunities for improving the computational efficiency of systems. In this work, we present an explorative study on 'model cascading', a simple technique that utilizes a collection of models of varying capacities to accurately yet efficiently output predictions. Through comprehensive experiments in multiple task settings that differ in the number of models available for cascading (K value), we show tha"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.05528","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.05528/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2210.05528","created_at":"2026-07-05T05:05:17.680415+00:00"},{"alias_kind":"arxiv_version","alias_value":"2210.05528v1","created_at":"2026-07-05T05:05:17.680415+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.05528","created_at":"2026-07-05T05:05:17.680415+00:00"},{"alias_kind":"pith_short_12","alias_value":"4KFBPBKNHWJO","created_at":"2026-07-05T05:05:17.680415+00:00"},{"alias_kind":"pith_short_16","alias_value":"4KFBPBKNHWJO4E46","created_at":"2026-07-05T05:05:17.680415+00:00"},{"alias_kind":"pith_short_8","alias_value":"4KFBPBKN","created_at":"2026-07-05T05:05:17.680415+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.26940","citing_title":"Select to Think: Unlocking SLM Potential with Local Sufficiency","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2409.11022","citing_title":"DynamicNER: A Dynamic, Multilingual, and Fine-Grained Dataset for LLM-based Named Entity Recognition","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2410.15761","citing_title":"Optimal Query Allocation in Extractive QA with LLMs: A Learning-to-Defer Framework with Theoretical Guarantees","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2502.18036","citing_title":"Harnessing Multiple Large Language Models: A Survey on LLM Ensemble","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26940","citing_title":"Select to Think: Unlocking SLM Potential with Local Sufficiency","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4KFBPBKNHWJO4E46DZBDPZBVLX","json":"https://pith.science/pith/4KFBPBKNHWJO4E46DZBDPZBVLX.json","graph_json":"https://pith.science/api/pith-number/4KFBPBKNHWJO4E46DZBDPZBVLX/graph.json","events_json":"https://pith.science/api/pith-number/4KFBPBKNHWJO4E46DZBDPZBVLX/events.json","paper":"https://pith.science/paper/4KFBPBKN"},"agent_actions":{"view_html":"https://pith.science/pith/4KFBPBKNHWJO4E46DZBDPZBVLX","download_json":"https://pith.science/pith/4KFBPBKNHWJO4E46DZBDPZBVLX.json","view_paper":"https://pith.science/paper/4KFBPBKN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2210.05528&json=true","fetch_graph":"https://pith.science/api/pith-number/4KFBPBKNHWJO4E46DZBDPZBVLX/graph.json","fetch_events":"https://pith.science/api/pith-number/4KFBPBKNHWJO4E46DZBDPZBVLX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4KFBPBKNHWJO4E46DZBDPZBVLX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4KFBPBKNHWJO4E46DZBDPZBVLX/action/storage_attestation","attest_author":"https://pith.science/pith/4KFBPBKNHWJO4E46DZBDPZBVLX/action/author_attestation","sign_citation":"https://pith.science/pith/4KFBPBKNHWJO4E46DZBDPZBVLX/action/citation_signature","submit_replication":"https://pith.science/pith/4KFBPBKNHWJO4E46DZBDPZBVLX/action/replication_record"}},"created_at":"2026-07-05T05:05:17.680415+00:00","updated_at":"2026-07-05T05:05:17.680415+00:00"}