{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:AM5BO454CCHEBUXU3FD4PXO4CG","short_pith_number":"pith:AM5BO454","schema_version":"1.0","canonical_sha256":"033a1773bc108e40d2f4d947c7dddc11aebdaaf8764a9bed17e359c3861aaf89","source":{"kind":"arxiv","id":"2503.08415","version":2},"attestation_state":"computed","paper":{"title":"TokenSim: Enabling Hardware and Software Exploration for Large Language Model Inference Systems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Feiyang Wu, Guoyang Duan, Junchi Wu, Ruihao Gong, Teng Ma, Tianle Xu, Yongqiang Yao, Youwei Zhuo, Zhuohang Bian","submitted_at":"2025-03-11T13:24:39Z","abstract_excerpt":"The increasing demand for large language model (LLM) serving has necessitated significant advancements in the optimization and profiling of LLM inference systems. As these models become integral to a wide range of applications, the need for efficient and scalable serving solutions has grown exponentially. This work introduces TokenSim, a comprehensive hardware and software exploration system designed specifically for LLM inference. TokenSim is characterized by its support for extensible system optimizations including scheduling and memory management. We validate the results with systems runnin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.08415","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.DC","submitted_at":"2025-03-11T13:24:39Z","cross_cats_sorted":[],"title_canon_sha256":"b53ddd69aad2239172f3a9bd139911a50bdd2668b6f2a7dea41e4d612a207bdf","abstract_canon_sha256":"9ac08dae98ec4e322a31664827639e1748d5724d0be7c88dea100bde65dd04c7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:34:43.341131Z","signature_b64":"I8jOKTA1KoPnusEYGGjP4WdWM8e/hR6LlgZPciBpXJprgQ+UynzxJ9qxPDr1bhN3eCPauvP5wSeo8fqavk8dAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"033a1773bc108e40d2f4d947c7dddc11aebdaaf8764a9bed17e359c3861aaf89","last_reissued_at":"2026-07-05T10:34:43.340602Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:34:43.340602Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TokenSim: Enabling Hardware and Software Exploration for Large Language Model Inference Systems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Feiyang Wu, Guoyang Duan, Junchi Wu, Ruihao Gong, Teng Ma, Tianle Xu, Yongqiang Yao, Youwei Zhuo, Zhuohang Bian","submitted_at":"2025-03-11T13:24:39Z","abstract_excerpt":"The increasing demand for large language model (LLM) serving has necessitated significant advancements in the optimization and profiling of LLM inference systems. As these models become integral to a wide range of applications, the need for efficient and scalable serving solutions has grown exponentially. This work introduces TokenSim, a comprehensive hardware and software exploration system designed specifically for LLM inference. TokenSim is characterized by its support for extensible system optimizations including scheduling and memory management. We validate the results with systems runnin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.08415","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.08415/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.08415","created_at":"2026-07-05T10:34:43.340671+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.08415v2","created_at":"2026-07-05T10:34:43.340671+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.08415","created_at":"2026-07-05T10:34:43.340671+00:00"},{"alias_kind":"pith_short_12","alias_value":"AM5BO454CCHE","created_at":"2026-07-05T10:34:43.340671+00:00"},{"alias_kind":"pith_short_16","alias_value":"AM5BO454CCHEBUXU","created_at":"2026-07-05T10:34:43.340671+00:00"},{"alias_kind":"pith_short_8","alias_value":"AM5BO454","created_at":"2026-07-05T10:34:43.340671+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28565","citing_title":"KernelSight-LM: A Kernel-Level LLM Inference Simulator","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28565","citing_title":"KernelSight-LM: A Kernel-Level LLM Inference Simulator","ref_index":50,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AM5BO454CCHEBUXU3FD4PXO4CG","json":"https://pith.science/pith/AM5BO454CCHEBUXU3FD4PXO4CG.json","graph_json":"https://pith.science/api/pith-number/AM5BO454CCHEBUXU3FD4PXO4CG/graph.json","events_json":"https://pith.science/api/pith-number/AM5BO454CCHEBUXU3FD4PXO4CG/events.json","paper":"https://pith.science/paper/AM5BO454"},"agent_actions":{"view_html":"https://pith.science/pith/AM5BO454CCHEBUXU3FD4PXO4CG","download_json":"https://pith.science/pith/AM5BO454CCHEBUXU3FD4PXO4CG.json","view_paper":"https://pith.science/paper/AM5BO454","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.08415&json=true","fetch_graph":"https://pith.science/api/pith-number/AM5BO454CCHEBUXU3FD4PXO4CG/graph.json","fetch_events":"https://pith.science/api/pith-number/AM5BO454CCHEBUXU3FD4PXO4CG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AM5BO454CCHEBUXU3FD4PXO4CG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AM5BO454CCHEBUXU3FD4PXO4CG/action/storage_attestation","attest_author":"https://pith.science/pith/AM5BO454CCHEBUXU3FD4PXO4CG/action/author_attestation","sign_citation":"https://pith.science/pith/AM5BO454CCHEBUXU3FD4PXO4CG/action/citation_signature","submit_replication":"https://pith.science/pith/AM5BO454CCHEBUXU3FD4PXO4CG/action/replication_record"}},"created_at":"2026-07-05T10:34:43.340671+00:00","updated_at":"2026-07-05T10:34:43.340671+00:00"}