{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:UZHPUBPZWMFGYPYR227MJLFCK4","short_pith_number":"pith:UZHPUBPZ","schema_version":"1.0","canonical_sha256":"a64efa05f9b30a6c3f11d6bec4aca257120b62509bb7953afae2ed1d68526355","source":{"kind":"arxiv","id":"2406.12375","version":1},"attestation_state":"computed","paper":{"title":"GW-MoE: Resolving Uncertainty in MoE Router with Global Workspace Theory","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Hang Zhao, Haoze Wu, Jie Fu, Zihan Qiu, Zili Wang","submitted_at":"2024-06-18T08:03:51Z","abstract_excerpt":"Mixture-of-Experts (MoE) has been demonstrated as an efficient method to scale up models. By dynamically and sparsely selecting activated experts, MoE can effectively reduce computational costs. Despite the success, we observe that many tokens in the MoE models have uncertain routing results. These tokens have nearly equal scores for choosing each expert, and we demonstrate that this uncertainty can lead to incorrect selections. Inspired by the Global Workspace Theory (GWT), we propose a new fine-tuning method, GW-MoE, to address this issue. The core idea is to broadcast the uncertain tokens a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.12375","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-06-18T08:03:51Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"9a188efee4f932373e15b7ec8625ad279e07c40174a6a3742b002debafd03260","abstract_canon_sha256":"eaac3ef102563187f2862d473ffd6e9b406d6a7796c1db9a9922c2b16773a021"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:33:45.075767Z","signature_b64":"jnF5apV66Y1kx9hUANYPerMlwyD0DY+TzKyuhuzw9FyYHO3o/t5WL0xbCVGojteI60Wpi+TdKfIo/fnCy6+iAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a64efa05f9b30a6c3f11d6bec4aca257120b62509bb7953afae2ed1d68526355","last_reissued_at":"2026-07-05T08:33:45.075351Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:33:45.075351Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GW-MoE: Resolving Uncertainty in MoE Router with Global Workspace Theory","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Hang Zhao, Haoze Wu, Jie Fu, Zihan Qiu, Zili Wang","submitted_at":"2024-06-18T08:03:51Z","abstract_excerpt":"Mixture-of-Experts (MoE) has been demonstrated as an efficient method to scale up models. By dynamically and sparsely selecting activated experts, MoE can effectively reduce computational costs. Despite the success, we observe that many tokens in the MoE models have uncertain routing results. These tokens have nearly equal scores for choosing each expert, and we demonstrate that this uncertainty can lead to incorrect selections. Inspired by the Global Workspace Theory (GWT), we propose a new fine-tuning method, GW-MoE, to address this issue. The core idea is to broadcast the uncertain tokens a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.12375","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.12375/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.12375","created_at":"2026-07-05T08:33:45.075409+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.12375v1","created_at":"2026-07-05T08:33:45.075409+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.12375","created_at":"2026-07-05T08:33:45.075409+00:00"},{"alias_kind":"pith_short_12","alias_value":"UZHPUBPZWMFG","created_at":"2026-07-05T08:33:45.075409+00:00"},{"alias_kind":"pith_short_16","alias_value":"UZHPUBPZWMFGYPYR","created_at":"2026-07-05T08:33:45.075409+00:00"},{"alias_kind":"pith_short_8","alias_value":"UZHPUBPZ","created_at":"2026-07-05T08:33:45.075409+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.11873","citing_title":"Demons in the Detail: On Implementing Load Balancing Loss for Training Specialized Mixture-of-Expert Models","ref_index":12,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UZHPUBPZWMFGYPYR227MJLFCK4","json":"https://pith.science/pith/UZHPUBPZWMFGYPYR227MJLFCK4.json","graph_json":"https://pith.science/api/pith-number/UZHPUBPZWMFGYPYR227MJLFCK4/graph.json","events_json":"https://pith.science/api/pith-number/UZHPUBPZWMFGYPYR227MJLFCK4/events.json","paper":"https://pith.science/paper/UZHPUBPZ"},"agent_actions":{"view_html":"https://pith.science/pith/UZHPUBPZWMFGYPYR227MJLFCK4","download_json":"https://pith.science/pith/UZHPUBPZWMFGYPYR227MJLFCK4.json","view_paper":"https://pith.science/paper/UZHPUBPZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.12375&json=true","fetch_graph":"https://pith.science/api/pith-number/UZHPUBPZWMFGYPYR227MJLFCK4/graph.json","fetch_events":"https://pith.science/api/pith-number/UZHPUBPZWMFGYPYR227MJLFCK4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UZHPUBPZWMFGYPYR227MJLFCK4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UZHPUBPZWMFGYPYR227MJLFCK4/action/storage_attestation","attest_author":"https://pith.science/pith/UZHPUBPZWMFGYPYR227MJLFCK4/action/author_attestation","sign_citation":"https://pith.science/pith/UZHPUBPZWMFGYPYR227MJLFCK4/action/citation_signature","submit_replication":"https://pith.science/pith/UZHPUBPZWMFGYPYR227MJLFCK4/action/replication_record"}},"created_at":"2026-07-05T08:33:45.075409+00:00","updated_at":"2026-07-05T08:33:45.075409+00:00"}