{"url":"/method/moe","slug":"moe","name":"MoE","full_name":"Mixture of Experts","full_name_withheld":false,"description_markdown":null,"description_state":"absent","introduced_year":null,"introduced_by":{"title":"Equipping Computational Pathology Systems with Artifact Processing Pipelines: A Showcase for Computation and Performance Trade-offs","paper":"/paper/equipping-computational-pathology-systems","first_author":"Neel Kanwal","n_authors":9,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/equipping-computational-pathology-systems"},"source":{"url":"https://arxiv.org/abs/2403.07743v3","title":"Equipping Computational Pathology Systems with Artifact Processing Pipelines: A Showcase for Computation and Performance Trade-offs","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"General","area_id":"general","collection":"Ensembling","url":"/methods/category/ensembling","pwc_aliases":[]}],"n_papers_tagged":366,"archive_num_papers":366,"papers_newest_first":[{"paper":null,"title":"Mixture of Experts in Large Language Models","date":"2025-07-15","arxiv_id":"2507.11181","n_code_links":0,"syntology":null},{"paper":"/paper/mofe-time-mixture-of-frequency-domain-experts","title":"MoFE-Time: Mixture of Frequency Domain Experts for Time-Series Forecasting Models","date":"2025-07-09","arxiv_id":"2507.06502","n_code_links":1,"syntology":null},{"paper":"/paper/growing-transformers-modular-composition-and","title":"Growing Transformers: Modular Composition and Layer-wise Expansion on a Frozen Substrate","date":"2025-07-08","arxiv_id":"2507.07129","n_code_links":1,"syntology":null},{"paper":null,"title":"Speech Quality Assessment Model Based on Mixture of Experts: System-Level Performance Enhancement and Utterance-Level Challenge Analysis","date":"2025-07-08","arxiv_id":"2507.06116","n_code_links":0,"syntology":null},{"paper":"/paper/learning-robust-stereo-matching-in-the-wild","title":"Learning Robust Stereo Matching in the Wild with Selective Mixture-of-Experts","date":"2025-07-07","arxiv_id":"2507.04631","n_code_links":1,"syntology":null},{"paper":null,"title":"Sub-MoE: Efficient Mixture-of-Expert LLMs Compression via Subspace Expert Merging","date":"2025-06-29","arxiv_id":"2506.23266","n_code_links":0,"syntology":null},{"paper":"/paper/latent-prototype-routing-achieving-near","title":"Latent Prototype Routing: Achieving Near-Perfect Load Balancing in Mixture-of-Experts","date":"2025-06-26","arxiv_id":"2506.21328","n_code_links":1,"syntology":null},{"paper":null,"title":"SAFEx: Analyzing Vulnerabilities of MoE-Based LLMs via Stable Safety-critical Expert Identification","date":"2025-06-20","arxiv_id":"2506.17368","n_code_links":0,"syntology":null},{"paper":null,"title":"Less is More: Undertraining Experts Improves Model Upcycling","date":"2025-06-17","arxiv_id":"2506.14126","n_code_links":0,"syntology":null},{"paper":null,"title":"LoRA-Mixer: Coordinate Modular LoRA Experts Through Serial Attention Routing","date":"2025-06-17","arxiv_id":"2507.00029","n_code_links":0,"syntology":null},{"paper":null,"title":"Ring-lite: Scalable Reasoning via C3PO-Stabilized Reinforcement Learning for LLMs","date":"2025-06-17","arxiv_id":"2506.14731","n_code_links":0,"syntology":null},{"paper":null,"title":"Utility-Driven Speculative Decoding for Mixture-of-Experts","date":"2025-06-17","arxiv_id":"2506.20675","n_code_links":0,"syntology":null},{"paper":null,"title":"EAQuant: Enhancing Post-Training Quantization for MoE Models via Expert-Aware Optimization","date":"2025-06-16","arxiv_id":"2506.13329","n_code_links":0,"syntology":null},{"paper":null,"title":"Serving Large Language Models on Huawei CloudMatrix384","date":"2025-06-15","arxiv_id":"2506.12708","n_code_links":0,"syntology":null},{"paper":"/paper/ming-omni-a-unified-multimodal-model-for","title":"Ming-Omni: A Unified Multimodal Model for Perception and Generation","date":"2025-06-11","arxiv_id":"2506.09344","n_code_links":1,"syntology":{"ran":4,"of":14,"unverified":10,"pointer_only":0}},{"paper":null,"title":"M2Restore: Mixture-of-Experts-based Mamba-CNN Fusion Framework for All-in-One Image Restoration","date":"2025-06-09","arxiv_id":"2506.07814","n_code_links":0,"syntology":null},{"paper":null,"title":"MoE-GPS: Guidlines for Prediction Strategy for Dynamic Expert Duplication in MoE Load Balancing","date":"2025-06-09","arxiv_id":"2506.07366","n_code_links":0,"syntology":null},{"paper":null,"title":"SMAR: Soft Modality-Aware Routing Strategy for MoE-based Multimodal Large Language Models Preserving Language Capabilities","date":"2025-06-06","arxiv_id":"2506.06406","n_code_links":0,"syntology":null},{"paper":"/paper/flashdmoe-fast-distributed-moe-in-a-single","title":"FlashDMoE: Fast Distributed MoE in a Single Kernel","date":"2025-06-05","arxiv_id":"2506.04667","n_code_links":2,"syntology":{"ran":2,"of":7,"unverified":5,"pointer_only":0}},{"paper":null,"title":"Out-of-Distribution Graph Models Merging","date":"2025-06-04","arxiv_id":"2506.03674","n_code_links":0,"syntology":null},{"paper":null,"title":"Decoding Knowledge Attribution in Mixture-of-Experts: A Framework of Basic-Refinement Collaboration and Efficiency Analysis","date":"2025-05-30","arxiv_id":"2505.24593","n_code_links":0,"syntology":null},{"paper":"/paper/mastering-massive-multi-task-reinforcement","title":"Mastering Massive Multi-Task Reinforcement Learning via Mixture-of-Expert Decision Transformer","date":"2025-05-30","arxiv_id":"2505.24378","n_code_links":1,"syntology":null},{"paper":null,"title":"Mixture-of-Experts for Personalized and Semantic-Aware Next Location Prediction","date":"2025-05-30","arxiv_id":"2505.24597","n_code_links":0,"syntology":null},{"paper":null,"title":"On the Expressive Power of Mixture-of-Experts for Structured Complex Tasks","date":"2025-05-30","arxiv_id":"2505.24205","n_code_links":0,"syntology":null},{"paper":null,"title":"Advancing Expert Specialization for Better MoE","date":"2025-05-28","arxiv_id":"2505.22323","n_code_links":0,"syntology":null},{"paper":null,"title":"EvoMoE: Expert Evolution in Mixture of Experts for Multimodal Large Language Models","date":"2025-05-28","arxiv_id":"2505.23830","n_code_links":0,"syntology":null},{"paper":"/paper/hidream-i1-a-high-efficient-image-generative","title":"HiDream-I1: A High-Efficient Image Generative Foundation Model with Sparse Diffusion Transformer","date":"2025-05-28","arxiv_id":"2505.22705","n_code_links":2,"syntology":{"ran":2,"of":3,"unverified":1,"pointer_only":0}},{"paper":"/paper/flame-moe-a-transparent-end-to-end-research","title":"FLAME-MoE: A Transparent End-to-End Research Platform for Mixture-of-Experts Language Models","date":"2025-05-26","arxiv_id":"2505.20225","n_code_links":1,"syntology":null},{"paper":null,"title":"MoESD: Unveil Speculative Decoding's Potential for Accelerating Sparse MoE","date":"2025-05-26","arxiv_id":"2505.19645","n_code_links":0,"syntology":null},{"paper":null,"title":"Mosaic: Data-Free Knowledge Distillation via Mixture-of-Experts for Heterogeneous Distributed Environments","date":"2025-05-26","arxiv_id":"2505.19699","n_code_links":0,"syntology":null}],"papers_shown":30,"tasks":[{"task":"/task/mixture-of-experts","name":"Mixture-of-Experts","papers":312},{"task":null,"name":"GPU","papers":38},{"task":"/task/language-modelling","name":"Language Modelling","papers":34},{"task":"/task/computational-efficiency","name":"Computational Efficiency","papers":30},{"task":"/task/language-modeling","name":"Language Modeling","papers":29},{"task":"/task/large-language-model","name":"Large Language Model","papers":16},{"task":"/task/quantization","name":"Quantization","papers":14},{"task":null,"name":"CPU","papers":13},{"task":"/task/multi-task-learning","name":"Multi-Task Learning","papers":13},{"task":"/task/image-classification","name":"Image Classification","papers":12},{"task":"/task/parameter-efficient-fine-tuning","name":"parameter-efficient fine-tuning","papers":12},{"task":"/task/scheduling","name":"Scheduling","papers":11},{"task":"/task/mmlu","name":"MMLU","papers":10},{"task":"/task/question-answering","name":"Question Answering","papers":10},{"task":"/task/decoder","name":"Decoder","papers":9},{"task":"/task/diversity","name":"Diversity","papers":9},{"task":"/task/image-classification","name":"image-classification","papers":9},{"task":"/task/continual-learning","name":"Continual Learning","papers":7},{"task":"/task/math","name":"Math","papers":7},{"task":"/task/retrieval","name":"Retrieval","papers":7}],"tasks_shown":20,"n_tasks":254,"usage_by_year":[{"year":"2022","papers":1},{"year":"2024","papers":206},{"year":"2025","papers":159}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/moe"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}