{"url":"/task/mixture-of-experts","name":"Mixture-of-Experts","slug":"mixture-of-experts","description_markdown":null,"categories":[],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"derived"},"counts":{"papers_tagged":1312,"papers_with_code":516,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":0,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":516,"tagged_in_all":1312,"items":[{"url":"/paper/distilling-the-knowledge-in-a-neural-network","title":"Distilling the Knowledge in a Neural Network","date":"2015-03-09","arxiv_id":"1503.02531","repositories_listed":64,"syntology":{"n":37,"n_ran":15,"n_unverified":22,"n_pointer_only":10}},{"url":"/paper/modeling-task-relationships-in-multi-task","title":"Modeling Task Relationships in Multi-task Learning with Multi-gate Mixture-of-Experts","date":"2018-07-19","arxiv_id":null,"repositories_listed":11,"syntology":null},{"url":"/paper/no-language-left-behind-scaling-human-1","title":"No Language Left Behind: Scaling Human-Centered Machine Translation","date":"2022-07-11","arxiv_id":"2207.04672","repositories_listed":9,"syntology":{"n":10,"n_ran":0,"n_unverified":10,"n_pointer_only":0}},{"url":"/paper/gated-multimodal-units-for-information-fusion","title":"Gated Multimodal Units for Information Fusion","date":"2017-02-07","arxiv_id":"1702.01992","repositories_listed":9,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/switch-transformers-scaling-to-trillion","title":"Switch Transformers: Scaling to Trillion Parameter Models with Simple and Efficient Sparsity","date":"2021-01-11","arxiv_id":"2101.03961","repositories_listed":8,"syntology":{"n":13,"n_ran":7,"n_unverified":6,"n_pointer_only":8}},{"url":"/paper/qwen2-5-technical-report","title":"Qwen2.5 Technical Report","date":"2024-12-19","arxiv_id":"2412.15115","repositories_listed":6,"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/qwen2-technical-report","title":"Qwen2 Technical Report","date":"2024-07-15","arxiv_id":"2407.10671","repositories_listed":6,"syntology":null},{"url":"/paper/mixtral-of-experts","title":"Mixtral of Experts","date":"2024-01-08","arxiv_id":"2401.04088","repositories_listed":6,"syntology":{"n":5,"n_ran":5,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/deepseek-v2-a-strong-economical-and-efficient","title":"DeepSeek-V2: A Strong, Economical, and Efficient Mixture-of-Experts Language Model","date":"2024-05-07","arxiv_id":"2405.04434","repositories_listed":5,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/jetmoe-reaching-llama2-performance-with-0-1m","title":"JetMoE: Reaching Llama2 Performance with 0.1M Dollars","date":"2024-04-11","arxiv_id":"2404.07413","repositories_listed":5,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/comet-fine-grained-computation-communication","title":"Comet: Fine-grained Computation-communication Overlapping for Mixture-of-Experts","date":"2025-02-27","arxiv_id":"2502.19811","repositories_listed":4,"syntology":{"n":20,"n_ran":1,"n_unverified":19,"n_pointer_only":0}},{"url":"/paper/deepseek-v3-technical-report","title":"DeepSeek-V3 Technical Report","date":"2024-12-27","arxiv_id":"2412.19437","repositories_listed":4,"syntology":{"n":7,"n_ran":0,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/unity-by-diversity-improved-representation","title":"Unity by Diversity: Improved Representation Learning in Multimodal VAEs","date":"2024-03-08","arxiv_id":"2403.05300","repositories_listed":4,"syntology":{"n":9,"n_ran":8,"n_unverified":1,"n_pointer_only":4}},{"url":"/paper/fast-feedforward-networks","title":"Fast Feedforward Networks","date":"2023-08-28","arxiv_id":"2308.14711","repositories_listed":4,"syntology":null},{"url":"/paper/long-tailed-visual-recognition-via-self","title":"Long-Tailed Visual Recognition via Self-Heterogeneous Integration with Knowledge Excavation","date":"2023-04-03","arxiv_id":"2304.01279","repositories_listed":4,"syntology":{"n":7,"n_ran":5,"n_unverified":2,"n_pointer_only":7}},{"url":"/paper/robust-federated-learning-by-mixture-of","title":"Robust Federated Learning by Mixture of Experts","date":"2021-04-23","arxiv_id":"2104.11700","repositories_listed":4,"syntology":null},{"url":"/paper/outrageously-large-neural-networks-the","title":"Outrageously Large Neural Networks: The Sparsely-Gated Mixture-of-Experts Layer","date":"2017-01-23","arxiv_id":"1701.06538","repositories_listed":4,"syntology":{"n":6,"n_ran":4,"n_unverified":2,"n_pointer_only":6}},{"url":"/paper/hunyuan-large-an-open-source-moe-model-with","title":"Hunyuan-Large: An Open-Source MoE Model with 52 Billion Activated Parameters by Tencent","date":"2024-11-04","arxiv_id":"2411.02265","repositories_listed":3,"syntology":null},{"url":"/paper/moh-multi-head-attention-as-mixture-of-head","title":"MoH: Multi-Head Attention as Mixture-of-Head Attention","date":"2024-10-15","arxiv_id":"2410.11842","repositories_listed":3,"syntology":{"n":18,"n_ran":14,"n_unverified":4,"n_pointer_only":3}},{"url":"/paper/moe-accelerating-mixture-of-experts-methods","title":"MoE++: Accelerating Mixture-of-Experts Methods with Zero-Computation Experts","date":"2024-10-09","arxiv_id":"2410.07348","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/jamba-a-hybrid-transformer-mamba-language","title":"Jamba: A Hybrid Transformer-Mamba Language Model","date":"2024-03-28","arxiv_id":"2403.19887","repositories_listed":3,"syntology":null},{"url":"/paper/scattered-mixture-of-experts-implementation","title":"Scattered Mixture-of-Experts Implementation","date":"2024-03-13","arxiv_id":"2403.08245","repositories_listed":3,"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/rethinking-llm-language-adaptation-a-case","title":"Rethinking LLM Language Adaptation: A Case Study on Chinese Mixtral","date":"2024-03-04","arxiv_id":"2403.01851","repositories_listed":3,"syntology":null},{"url":"/paper/moe-llava-mixture-of-experts-for-large-vision","title":"MoE-LLaVA: Mixture of Experts for Large Vision-Language Models","date":"2024-01-29","arxiv_id":"2401.15947","repositories_listed":3,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":2}},{"url":"/paper/megablocks-efficient-sparse-training-with","title":"MegaBlocks: Efficient Sparse Training with Mixture-of-Experts","date":"2022-11-29","arxiv_id":"2211.15841","repositories_listed":3,"syntology":{"n":6,"n_ran":1,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/designing-effective-sparse-expert-models","title":"ST-MoE: Designing Stable and Transferable Sparse Expert Models","date":"2022-02-17","arxiv_id":"2202.08906","repositories_listed":3,"syntology":{"n":5,"n_ran":5,"n_unverified":0,"n_pointer_only":5}},{"url":"/paper/deepspeed-moe-advancing-mixture-of-experts","title":"DeepSpeed-MoE: Advancing Mixture-of-Experts Inference and Training to Power Next-Generation AI Scale","date":"2022-01-14","arxiv_id":"2201.05596","repositories_listed":3,"syntology":null},{"url":"/paper/dselect-k-differentiable-selection-in-the","title":"DSelect-k: Differentiable Selection in the Mixture of Experts with Applications to Multi-Task Learning","date":"2021-06-07","arxiv_id":"2106.03760","repositories_listed":3,"syntology":{"n":9,"n_ran":1,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/variational-mixture-of-experts-autoencoders","title":"Variational Mixture-of-Experts Autoencoders for Multi-Modal Deep Generative Models","date":"2019-11-08","arxiv_id":"1911.03393","repositories_listed":3,"syntology":{"n":8,"n_ran":1,"n_unverified":7,"n_pointer_only":1}},{"url":"/paper/non-normal-mixtures-of-experts","title":"Non-Normal Mixtures of Experts","date":"2015-06-22","arxiv_id":"1506.06707","repositories_listed":3,"syntology":null}],"syntology_records":21,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}