{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FBOWBOYJIVPFO3PDJN4MJ77QSX","short_pith_number":"pith:FBOWBOYJ","schema_version":"1.0","canonical_sha256":"285d60bb09455e576de34b78c4fff095eb6a7f85f914ca3fe4e9c63f78fb83bc","source":{"kind":"arxiv","id":"2505.21411","version":2},"attestation_state":"computed","paper":{"title":"Pangu Pro MoE: Mixture of Grouped Experts for Efficient Sparsity","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Binfan Zheng, Can Chen, Dacheng Tao, Fangcheng Liu, Fei Mi, Hang Zhou, Hanting Chen, Hui Zang, Jinpeng Li, Kai Han, Peifeng Qin, Ruiming Tang, Wei Guo, Xianzhi Yu, Xiaojun Meng, Xiaosong Li, Xinghao Chen, Yaoyuan Wang, Yehui Tang, Youliang Yan, Yunhe Wang (and Other Contributors), Zhicheng Liu","submitted_at":"2025-05-27T16:40:21Z","abstract_excerpt":"The surgence of Mixture of Experts (MoE) in Large Language Models promises a small price of execution cost for a much larger model parameter count and learning capacity, because only a small fraction of parameters are activated for each input token. However, it is commonly observed that some experts are activated far more often than others, leading to system inefficiency when running the experts on different devices in parallel. Therefore, we introduce Mixture of Grouped Experts (MoGE), which groups the experts during selection and balances the expert workload better than MoE in nature. It con"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.21411","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-05-27T16:40:21Z","cross_cats_sorted":[],"title_canon_sha256":"0ef3bfbd40623b7f1881271ddbb3c19288462e1894d30dad9da56fe823f66dda","abstract_canon_sha256":"1bfc9b1b3e7d0dbd8b43785676ab6e3b6adbcd71ede476d0b69fbdfab794eea2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:10:59.074166Z","signature_b64":"DCwRE+HgYKW18xE5D6vVHI6rQWG2rrvLTrhxFwO+EM2Oa7rC6B0dtO69r8zFMCuQuar3x/bCWki94paGLzbEDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"285d60bb09455e576de34b78c4fff095eb6a7f85f914ca3fe4e9c63f78fb83bc","last_reissued_at":"2026-07-05T11:10:59.073690Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:10:59.073690Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Pangu Pro MoE: Mixture of Grouped Experts for Efficient Sparsity","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Binfan Zheng, Can Chen, Dacheng Tao, Fangcheng Liu, Fei Mi, Hang Zhou, Hanting Chen, Hui Zang, Jinpeng Li, Kai Han, Peifeng Qin, Ruiming Tang, Wei Guo, Xianzhi Yu, Xiaojun Meng, Xiaosong Li, Xinghao Chen, Yaoyuan Wang, Yehui Tang, Youliang Yan, Yunhe Wang (and Other Contributors), Zhicheng Liu","submitted_at":"2025-05-27T16:40:21Z","abstract_excerpt":"The surgence of Mixture of Experts (MoE) in Large Language Models promises a small price of execution cost for a much larger model parameter count and learning capacity, because only a small fraction of parameters are activated for each input token. However, it is commonly observed that some experts are activated far more often than others, leading to system inefficiency when running the experts on different devices in parallel. Therefore, we introduce Mixture of Grouped Experts (MoGE), which groups the experts during selection and balances the expert workload better than MoE in nature. It con"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.21411","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.21411/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.21411","created_at":"2026-07-05T11:10:59.073746+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.21411v2","created_at":"2026-07-05T11:10:59.073746+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.21411","created_at":"2026-07-05T11:10:59.073746+00:00"},{"alias_kind":"pith_short_12","alias_value":"FBOWBOYJIVPF","created_at":"2026-07-05T11:10:59.073746+00:00"},{"alias_kind":"pith_short_16","alias_value":"FBOWBOYJIVPFO3PD","created_at":"2026-07-05T11:10:59.073746+00:00"},{"alias_kind":"pith_short_8","alias_value":"FBOWBOYJ","created_at":"2026-07-05T11:10:59.073746+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03465","citing_title":"Rethinking the Role of Tensor Decompositions in Post-Training LLM Compression","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23893","citing_title":"Complete-muE: Optimal Hyperparameter Transfer and Scaling for MoE Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23294","citing_title":"NASiC: 3D NAND-based CAM-Selected Multibit CIM Architecture for Efficient On-Device Mixture-of-Experts LLM Inference","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2507.03014","citing_title":"Intrinsic Fingerprint of LLMs: Continue Training is NOT All You Need to Steal A Model!","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08292","citing_title":"Hierarchical Mixture-of-Experts with Two-Stage Optimization","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23996","citing_title":"SMoES: Soft Modality-Guided Expert Specialization in MoE-VLMs","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02946","citing_title":"RouteHijack: Routing-Aware Attack on Mixture-of-Experts LLMs","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04952","citing_title":"Adaptive Inverted-Index Routing for Granular Mixtures-of-Experts","ref_index":46,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FBOWBOYJIVPFO3PDJN4MJ77QSX","json":"https://pith.science/pith/FBOWBOYJIVPFO3PDJN4MJ77QSX.json","graph_json":"https://pith.science/api/pith-number/FBOWBOYJIVPFO3PDJN4MJ77QSX/graph.json","events_json":"https://pith.science/api/pith-number/FBOWBOYJIVPFO3PDJN4MJ77QSX/events.json","paper":"https://pith.science/paper/FBOWBOYJ"},"agent_actions":{"view_html":"https://pith.science/pith/FBOWBOYJIVPFO3PDJN4MJ77QSX","download_json":"https://pith.science/pith/FBOWBOYJIVPFO3PDJN4MJ77QSX.json","view_paper":"https://pith.science/paper/FBOWBOYJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.21411&json=true","fetch_graph":"https://pith.science/api/pith-number/FBOWBOYJIVPFO3PDJN4MJ77QSX/graph.json","fetch_events":"https://pith.science/api/pith-number/FBOWBOYJIVPFO3PDJN4MJ77QSX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FBOWBOYJIVPFO3PDJN4MJ77QSX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FBOWBOYJIVPFO3PDJN4MJ77QSX/action/storage_attestation","attest_author":"https://pith.science/pith/FBOWBOYJIVPFO3PDJN4MJ77QSX/action/author_attestation","sign_citation":"https://pith.science/pith/FBOWBOYJIVPFO3PDJN4MJ77QSX/action/citation_signature","submit_replication":"https://pith.science/pith/FBOWBOYJIVPFO3PDJN4MJ77QSX/action/replication_record"}},"created_at":"2026-07-05T11:10:59.073746+00:00","updated_at":"2026-07-05T11:10:59.073746+00:00"}