{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:QIBR76MS7UCIJAUVD3SUSCB2W3","short_pith_number":"pith:QIBR76MS","schema_version":"1.0","canonical_sha256":"82031ff992fd048482951ee549083ab6e2e1d6ee92237f77f3f4f9bc91890de1","source":{"kind":"arxiv","id":"2206.03382","version":2},"attestation_state":"computed","paper":{"title":"Tutel: Adaptive Mixture-of-Experts at Scale","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CV"],"primary_cat":"cs.DC","authors_text":"Changho Hwang, Fan Yang, Han Hu, Jithin Jose, Joe Chau, Mao Yang, Peng Cheng, Prabhat Ram, Rafael Salas, Wei Cui, Yifan Xiong, Yongqiang Xiong, Ze Liu, Zilong Wang, Ziyue Yang","submitted_at":"2022-06-07T15:20:20Z","abstract_excerpt":"Sparsely-gated mixture-of-experts (MoE) has been widely adopted to scale deep learning models to trillion-plus parameters with fixed computational cost. The algorithmic performance of MoE relies on its token routing mechanism that forwards each input token to the right sub-models or experts. While token routing dynamically determines the amount of expert workload at runtime, existing systems suffer inefficient computation due to their static execution, namely static parallelism and pipelining, which does not adapt to the dynamic workload. We present Flex, a highly scalable stack design and imp"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2206.03382","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.DC","submitted_at":"2022-06-07T15:20:20Z","cross_cats_sorted":["cs.CL","cs.CV"],"title_canon_sha256":"d4037af39324da8b3b0b321a5bc9d09f9adfb398da550a78e98f339bebe93c21","abstract_canon_sha256":"aa6ac843c5d1fa8ee7021000afd354eb7fba80c925bec7385de3f650b8d4fbb3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:17:25.783338Z","signature_b64":"TaQvpFBU0c9xef5P/EIEP18le6MNufA2B1jTdX/y/wqQFaie67yor8eyN8Ad1g4DGJiF/d9Cf+OAjTBXrbJXDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"82031ff992fd048482951ee549083ab6e2e1d6ee92237f77f3f4f9bc91890de1","last_reissued_at":"2026-07-05T06:17:25.782864Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:17:25.782864Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Tutel: Adaptive Mixture-of-Experts at Scale","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CV"],"primary_cat":"cs.DC","authors_text":"Changho Hwang, Fan Yang, Han Hu, Jithin Jose, Joe Chau, Mao Yang, Peng Cheng, Prabhat Ram, Rafael Salas, Wei Cui, Yifan Xiong, Yongqiang Xiong, Ze Liu, Zilong Wang, Ziyue Yang","submitted_at":"2022-06-07T15:20:20Z","abstract_excerpt":"Sparsely-gated mixture-of-experts (MoE) has been widely adopted to scale deep learning models to trillion-plus parameters with fixed computational cost. The algorithmic performance of MoE relies on its token routing mechanism that forwards each input token to the right sub-models or experts. While token routing dynamically determines the amount of expert workload at runtime, existing systems suffer inefficient computation due to their static execution, namely static parallelism and pipelining, which does not adapt to the dynamic workload. We present Flex, a highly scalable stack design and imp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.03382","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2206.03382/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2206.03382","created_at":"2026-07-05T06:17:25.782927+00:00"},{"alias_kind":"arxiv_version","alias_value":"2206.03382v2","created_at":"2026-07-05T06:17:25.782927+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.03382","created_at":"2026-07-05T06:17:25.782927+00:00"},{"alias_kind":"pith_short_12","alias_value":"QIBR76MS7UCI","created_at":"2026-07-05T06:17:25.782927+00:00"},{"alias_kind":"pith_short_16","alias_value":"QIBR76MS7UCIJAUV","created_at":"2026-07-05T06:17:25.782927+00:00"},{"alias_kind":"pith_short_8","alias_value":"QIBR76MS","created_at":"2026-07-05T06:17:25.782927+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.20982","citing_title":"Diagnosing Overhead in Dispatch Operations: Cross-architecture Observatory","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2508.21613","citing_title":"Chameleon: Adaptive Fault Tolerance for Distributed Training via Real-time Policy Selection","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11005","citing_title":"DisagMoE: Computation-Communication overlapped MoE Training via Disaggregated AF-Pipe Parallelism","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10670","citing_title":"Surviving Partial Rank Failures in Wide Expert-Parallel MoE Inference","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08292","citing_title":"Hierarchical Mixture-of-Experts with Two-Stage Optimization","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04750","citing_title":"DeepStack: Scalable and Accurate Design Space Exploration for Distributed 3D-Stacked AI Accelerators","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05049","citing_title":"Piper: Efficient Large-Scale MoE Training via Resource Modeling and Pipelined Hybrid Parallelism","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QIBR76MS7UCIJAUVD3SUSCB2W3","json":"https://pith.science/pith/QIBR76MS7UCIJAUVD3SUSCB2W3.json","graph_json":"https://pith.science/api/pith-number/QIBR76MS7UCIJAUVD3SUSCB2W3/graph.json","events_json":"https://pith.science/api/pith-number/QIBR76MS7UCIJAUVD3SUSCB2W3/events.json","paper":"https://pith.science/paper/QIBR76MS"},"agent_actions":{"view_html":"https://pith.science/pith/QIBR76MS7UCIJAUVD3SUSCB2W3","download_json":"https://pith.science/pith/QIBR76MS7UCIJAUVD3SUSCB2W3.json","view_paper":"https://pith.science/paper/QIBR76MS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2206.03382&json=true","fetch_graph":"https://pith.science/api/pith-number/QIBR76MS7UCIJAUVD3SUSCB2W3/graph.json","fetch_events":"https://pith.science/api/pith-number/QIBR76MS7UCIJAUVD3SUSCB2W3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QIBR76MS7UCIJAUVD3SUSCB2W3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QIBR76MS7UCIJAUVD3SUSCB2W3/action/storage_attestation","attest_author":"https://pith.science/pith/QIBR76MS7UCIJAUVD3SUSCB2W3/action/author_attestation","sign_citation":"https://pith.science/pith/QIBR76MS7UCIJAUVD3SUSCB2W3/action/citation_signature","submit_replication":"https://pith.science/pith/QIBR76MS7UCIJAUVD3SUSCB2W3/action/replication_record"}},"created_at":"2026-07-05T06:17:25.782927+00:00","updated_at":"2026-07-05T06:17:25.782927+00:00"}