{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:KKDT3IA7IZHCK2VNFDABBTOZ7H","short_pith_number":"pith:KKDT3IA7","schema_version":"1.0","canonical_sha256":"52873da01f464e256aad28c010cdd9f9c89c782a27c74158d388529668d90439","source":{"kind":"arxiv","id":"2312.03863","version":4},"attestation_state":"computed","paper":{"title":"Efficient Large Language Models: A Survey","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Che Liu, Jiachen Liu, Mi Zhang, Mosharaf Chowdhury, Quanlu Zhang, Samiul Alam, Shen Yan, Xin Wang, Yi Zhu, Yu Zheng, Zhongnan Qu, Zhongwei Wan","submitted_at":"2023-12-06T19:18:42Z","abstract_excerpt":"Large Language Models (LLMs) have demonstrated remarkable capabilities in important tasks such as natural language understanding and language generation, and thus have the potential to make a substantial impact on our society. Such capabilities, however, come with the considerable resources they demand, highlighting the strong need to develop effective techniques for addressing their efficiency challenges. In this survey, we provide a systematic and comprehensive review of efficient LLMs research. We organize the literature in a taxonomy consisting of three main categories, covering distinct y"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.03863","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-12-06T19:18:42Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"2ea3e3ebc8ba53243c4b9ab7ce9fd82e7c5bc582f4f396cc120d6f24e160d065","abstract_canon_sha256":"e6fcf553001f8ef9d528497763f31c189ce0846a47f2bc08ddb4ce3b761e8ccd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:21:45.335200Z","signature_b64":"1KLCqVN6pZLhX7rtDv9uB6DXzwd3gfMYE3B07lkQ914MQbckXq6PWO1S61RZqYtbQdLtbN/wyF3CjPGrDS1FDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"52873da01f464e256aad28c010cdd9f9c89c782a27c74158d388529668d90439","last_reissued_at":"2026-07-05T08:21:45.334688Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:21:45.334688Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Efficient Large Language Models: A Survey","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Che Liu, Jiachen Liu, Mi Zhang, Mosharaf Chowdhury, Quanlu Zhang, Samiul Alam, Shen Yan, Xin Wang, Yi Zhu, Yu Zheng, Zhongnan Qu, Zhongwei Wan","submitted_at":"2023-12-06T19:18:42Z","abstract_excerpt":"Large Language Models (LLMs) have demonstrated remarkable capabilities in important tasks such as natural language understanding and language generation, and thus have the potential to make a substantial impact on our society. Such capabilities, however, come with the considerable resources they demand, highlighting the strong need to develop effective techniques for addressing their efficiency challenges. In this survey, we provide a systematic and comprehensive review of efficient LLMs research. We organize the literature in a taxonomy consisting of three main categories, covering distinct y"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.03863","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.03863/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.03863","created_at":"2026-07-05T08:21:45.334758+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.03863v4","created_at":"2026-07-05T08:21:45.334758+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.03863","created_at":"2026-07-05T08:21:45.334758+00:00"},{"alias_kind":"pith_short_12","alias_value":"KKDT3IA7IZHC","created_at":"2026-07-05T08:21:45.334758+00:00"},{"alias_kind":"pith_short_16","alias_value":"KKDT3IA7IZHCK2VN","created_at":"2026-07-05T08:21:45.334758+00:00"},{"alias_kind":"pith_short_8","alias_value":"KKDT3IA7","created_at":"2026-07-05T08:21:45.334758+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":28,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08032","citing_title":"What to Keep, What to Forget: A Rate--Distortion View of Memory Compaction in LLMs and Agents","ref_index":117,"is_internal_anchor":true},{"citing_arxiv_id":"2606.27147","citing_title":"Safe Autoregressive Image Generation with Iterative Self-Improving Codebooks","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21238","citing_title":"Recency/Frequency Adaptive KV Caching for Large Language Model Serving","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10650","citing_title":"Dynamic Linear Attention","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07632","citing_title":"Evaluation of ML Resource Utilization Requires Model Life Cycle Assessment","ref_index":103,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08876","citing_title":"OTora: A Unified Red Teaming Framework for Reasoning-Level Denial-of-Service in LLM Agents","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31315","citing_title":"BlockPilot: Instance-Adaptive Policy Learning for Diffusion-based Speculative Decoding","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23078","citing_title":"GEMQ: Global Expert-Level Mixed-Precision Quantization for MoE LLMs","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2410.20791","citing_title":"From Cool Demos to Production-Ready FMware: Core Challenges and a Technology Roadmap","ref_index":101,"is_internal_anchor":false},{"citing_arxiv_id":"2505.16278","citing_title":"DriveMoE: Mixture-of-Experts for Vision-Language-Action Model in End-to-End Autonomous Driving","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07035","citing_title":"Unified Deployment-Aware Evaluation of Open Reasoning Language Models","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15104","citing_title":"From Text to Voice: A Reproducible and Verifiable Framework for Evaluating Tool Calling LLM Agents","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18597","citing_title":"Latent Action Reparameterization for Efficient Agent Inference","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19394","citing_title":"EmbGen: Teaching with Reassembled Corpora","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2510.25977","citing_title":"NeuronMLP: Efficient LLM Inference via Singular Value Decomposition Compression and Tiling on AWS Trainium","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2602.01997","citing_title":"On the Limits of Layer Pruning for Generative Reasoning in Large Language Models","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2602.15089","citing_title":"Triplet Feature Fusion for Equipment Anomaly Prediction : An Open-Source Methodology Using Small Foundation Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2404.13501","citing_title":"A Survey on the Memory Mechanism of Large Language Model based Agents","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2404.14294","citing_title":"A Survey on Efficient Inference for Large Language Models","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2603.25383","citing_title":"CLIP-RD: Relative Distillation for Efficient CLIP Knowledge Distillation","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13405","citing_title":"When is Warmstarting Effective for Scaling Language Models?","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08876","citing_title":"OTora: A Unified Red Teaming Framework for Reasoning-Level Denial-of-Service in LLM Agents","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04738","citing_title":"OSAQ: Outlier Self-Absorption for Accurate Low-bit LLM Quantization","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06548","citing_title":"Continuous Latent Diffusion Language Model","ref_index":96,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07035","citing_title":"Unified Deployment-Aware Evaluation of Open Reasoning Language Models","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KKDT3IA7IZHCK2VNFDABBTOZ7H","json":"https://pith.science/pith/KKDT3IA7IZHCK2VNFDABBTOZ7H.json","graph_json":"https://pith.science/api/pith-number/KKDT3IA7IZHCK2VNFDABBTOZ7H/graph.json","events_json":"https://pith.science/api/pith-number/KKDT3IA7IZHCK2VNFDABBTOZ7H/events.json","paper":"https://pith.science/paper/KKDT3IA7"},"agent_actions":{"view_html":"https://pith.science/pith/KKDT3IA7IZHCK2VNFDABBTOZ7H","download_json":"https://pith.science/pith/KKDT3IA7IZHCK2VNFDABBTOZ7H.json","view_paper":"https://pith.science/paper/KKDT3IA7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.03863&json=true","fetch_graph":"https://pith.science/api/pith-number/KKDT3IA7IZHCK2VNFDABBTOZ7H/graph.json","fetch_events":"https://pith.science/api/pith-number/KKDT3IA7IZHCK2VNFDABBTOZ7H/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KKDT3IA7IZHCK2VNFDABBTOZ7H/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KKDT3IA7IZHCK2VNFDABBTOZ7H/action/storage_attestation","attest_author":"https://pith.science/pith/KKDT3IA7IZHCK2VNFDABBTOZ7H/action/author_attestation","sign_citation":"https://pith.science/pith/KKDT3IA7IZHCK2VNFDABBTOZ7H/action/citation_signature","submit_replication":"https://pith.science/pith/KKDT3IA7IZHCK2VNFDABBTOZ7H/action/replication_record"}},"created_at":"2026-07-05T08:21:45.334758+00:00","updated_at":"2026-07-05T08:21:45.334758+00:00"}