{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/moe/papers/3","list_of":"/method/moe","method":"MoE","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":3,"pages_in_order":4,"rows_per_page":100,"rows":[201,300],"of":366,"counts":{"archive_papers_tagged":366,"with_a_code_link":142,"where_syntology_ran_a_sample":64,"not_listed_spam_title":0,"listed":366,"listed_where_code_ran":64,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":49,"every_run_a_failure_of_syntologys_instrument":15,"listed_with_a_run_with_no_instrument_failure":49,"listed_every_run_a_failure_of_syntologys_instrument":15,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/moe","prev":"/method/moe/papers/2","next":"/method/moe/papers/4","papers":[{"paper":"/paper/libmoe-a-library-for-comprehensive","slug":"libmoe-a-library-for-comprehensive","title":"LIBMoE: A Library for comprehensive benchmarking Mixture of Experts in Large Language Models","date":"2024-11-01","arxiv_id":"2411.00918","n_code_links":1,"syntology":null},{"paper":null,"slug":"moe-i-2-compressing-mixture-of-experts-models","title":"MoE-I$^2$: Compressing Mixture of Experts Models through Inter-Expert Pruning and Intra-Expert Low-Rank Decomposition","date":"2024-11-01","arxiv_id":"2411.01016","n_code_links":0,"syntology":null},{"paper":null,"slug":"stereo-talker-audio-driven-3d-human-synthesis","title":"Stereo-Talker: Audio-driven 3D Human Synthesis with Prior-Guided Mixture-of-Experts","date":"2024-10-31","arxiv_id":"2410.23836","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-and-effective-weight-ensembling","title":"Efficient and Effective Weight-Ensembling Mixture of Experts for Multi-Task Model Merging","date":"2024-10-29","arxiv_id":"2410.21804","n_code_links":0,"syntology":null},{"paper":null,"slug":"gumbel-nerf-representing-unseen-objects-as","title":"GUMBEL-NERF: Representing Unseen Objects as Part-Compositional Neural Radiance Fields","date":"2024-10-27","arxiv_id":"2410.20306","n_code_links":0,"syntology":null},{"paper":"/paper/dmt-hi-moe-based-hyperbolic-interpretable","slug":"dmt-hi-moe-based-hyperbolic-interpretable","title":"DMT-HI: MOE-based Hyperbolic Interpretable Deep Manifold Transformation for Unspervised Dimensionality Reduction","date":"2024-10-25","arxiv_id":"2410.19504","n_code_links":1,"syntology":null},{"paper":"/paper/hierarchical-mixture-of-experts-generalizable","slug":"hierarchical-mixture-of-experts-generalizable","title":"Hierarchical Mixture of Experts: Generalizable Learning for High-Level Synthesis","date":"2024-10-25","arxiv_id":"2410.19225","n_code_links":1,"syntology":null},{"paper":"/paper/read-me-refactorizing-llms-as-router","slug":"read-me-refactorizing-llms-as-router","title":"Read-ME: Refactorizing LLMs as Router-Decoupled Mixture of Experts with System Co-Design","date":"2024-10-24","arxiv_id":"2410.19123","n_code_links":1,"syntology":{"ran":11,"of":16,"n_ran_checked":10,"n_instrument":1,"unverified":5,"pointer_only":16,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":{"repos":["vita-group/read-me"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"expertflow-optimized-expert-activation-and","title":"ExpertFlow: Optimized Expert Activation and Token Allocation for Efficient Mixture-of-Experts Inference","date":"2024-10-23","arxiv_id":"2410.17954","n_code_links":0,"syntology":null},{"paper":null,"slug":"milora-efficient-mixture-of-low-rank","title":"MiLoRA: Efficient Mixture of Low-Rank Adaptation for Large Language Models Fine-tuning","date":"2024-10-23","arxiv_id":"2410.18035","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimizing-mixture-of-experts-inference-time","title":"Optimizing Mixture-of-Experts Inference Time Combining Model Deployment and Communication Scheduling","date":"2024-10-22","arxiv_id":"2410.17043","n_code_links":0,"syntology":null},{"paper":"/paper/cartesianmoe-boosting-knowledge-sharing-among","slug":"cartesianmoe-boosting-knowledge-sharing-among","title":"CartesianMoE: Boosting Knowledge Sharing among Experts via Cartesian Product Routing in Mixture-of-Experts","date":"2024-10-21","arxiv_id":"2410.16077","n_code_links":1,"syntology":null},{"paper":"/paper/generalizing-motion-planners-with-mixture-of","slug":"generalizing-motion-planners-with-mixture-of","title":"Generalizing Motion Planners with Mixture of Experts for Autonomous Driving","date":"2024-10-21","arxiv_id":"2410.15774","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tsinghua-mars-lab/statetransformer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"vimoe-an-empirical-study-of-designing-vision","title":"ViMoE: An Empirical Study of Designing Vision Mixture-of-Experts","date":"2024-10-21","arxiv_id":"2410.15732","n_code_links":0,"syntology":null},{"paper":"/paper/collaboratively-adding-new-knowledge-to-an","slug":"collaboratively-adding-new-knowledge-to-an","title":"Collaboratively adding new knowledge to an LLM","date":"2024-10-18","arxiv_id":"2410.14753","n_code_links":1,"syntology":null},{"paper":"/paper/momentumsmoe-integrating-momentum-into-sparse","slug":"momentumsmoe-integrating-momentum-into-sparse","title":"MomentumSMoE: Integrating Momentum into Sparse Mixture of Experts","date":"2024-10-18","arxiv_id":"2410.14574","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["rachtsy/momentumsmoe"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"eps-moe-expert-pipeline-scheduler-for-cost","title":"EPS-MoE: Expert Pipeline Scheduler for Cost-Efficient MoE Inference","date":"2024-10-16","arxiv_id":"2410.12247","n_code_links":0,"syntology":null},{"paper":null,"slug":"moe-pruner-pruning-mixture-of-experts-large","title":"MoE-Pruner: Pruning Mixture-of-Experts Large Language Model using the Hints from Its Router","date":"2024-10-15","arxiv_id":"2410.12013","n_code_links":0,"syntology":null},{"paper":null,"slug":"quadratic-gating-functions-in-mixture-of","title":"Quadratic Gating Functions in Mixture of Experts: A Statistical Insight","date":"2024-10-15","arxiv_id":"2410.11222","n_code_links":0,"syntology":null},{"paper":null,"slug":"ada-k-routing-boosting-the-efficiency-of-moe","title":"Ada-K Routing: Boosting the Efficiency of MoE-based LLMs","date":"2024-10-14","arxiv_id":"2410.10456","n_code_links":0,"syntology":null},{"paper":"/paper/alphalora-assigning-lora-experts-based-on","slug":"alphalora-assigning-lora-experts-based-on","title":"AlphaLoRA: Assigning LoRA Experts Based on Layer Training Quality","date":"2024-10-14","arxiv_id":"2410.10054","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["morelife2017/alphalora"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/efficiently-democratizing-medical-llms-for-50","slug":"efficiently-democratizing-medical-llms-for-50","title":"Efficiently Democratizing Medical LLMs for 50 Languages via a Mixture of Language Family Experts","date":"2024-10-14","arxiv_id":"2410.10626","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":0,"n_instrument":4,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["freedomintelligence/apollomoe"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"learning-to-ground-vlms-without-forgetting","title":"Learning to Ground VLMs without Forgetting","date":"2024-10-14","arxiv_id":"2410.10491","n_code_links":0,"syntology":null},{"paper":"/paper/moirai-moe-empowering-time-series-foundation","slug":"moirai-moe-empowering-time-series-foundation","title":"Moirai-MoE: Empowering Time Series Foundation Models with Sparse Mixture of Experts","date":"2024-10-14","arxiv_id":"2410.10469","n_code_links":1,"syntology":null},{"paper":"/paper/your-mixture-of-experts-llm-is-secretly-an","slug":"your-mixture-of-experts-llm-is-secretly-an","title":"Your Mixture-of-Experts LLM Is Secretly an Embedding Model For Free","date":"2024-10-14","arxiv_id":"2410.10814","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":1,"n_instrument":5,"unverified":0,"pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tianyi-lab/moe-embedding"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"at-moe-adaptive-task-planning-mixture-of","title":"AT-MoE: Adaptive Task-planning Mixture of Experts via LoRA Approach","date":"2024-10-12","arxiv_id":"2410.10896","n_code_links":0,"syntology":null},{"paper":"/paper/flex-moe-modeling-arbitrary-modality","slug":"flex-moe-modeling-arbitrary-modality","title":"Flex-MoE: Modeling Arbitrary Modality Combination via the Flexible Mixture-of-Experts","date":"2024-10-10","arxiv_id":"2410.08245","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["unites-lab/flex-moe"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"upcycling-large-language-models-into-mixture","title":"Upcycling Large Language Models into Mixture of Experts","date":"2024-10-10","arxiv_id":"2410.07524","n_code_links":0,"syntology":null},{"paper":"/paper/moe-accelerating-mixture-of-experts-methods","slug":"moe-accelerating-mixture-of-experts-methods","title":"MoE++: Accelerating Mixture-of-Experts Methods with Zero-Computation Experts","date":"2024-10-09","arxiv_id":"2410.07348","n_code_links":3,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["skyworkai/moe-plus-plus"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"paper":null,"slug":"scaling-laws-across-model-architectures-a","title":"Scaling Laws Across Model Architectures: A Comparative Analysis of Dense and MoE Models in Large Language Models","date":"2024-10-08","arxiv_id":"2410.05661","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-dynamic-approach-to-stock-price-prediction","title":"A Dynamic Approach to Stock Price Prediction: Comparing RNN and Mixture of Experts Models Across Different Volatility Profiles","date":"2024-10-04","arxiv_id":"2410.07234","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-the-benefit-of-activation-sparsity","slug":"exploring-the-benefit-of-activation-sparsity","title":"Exploring the Benefit of Activation Sparsity in Pre-training","date":"2024-10-04","arxiv_id":"2410.03440","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["thunlp/moefication"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"efficient-residual-learning-with-mixture-of","title":"Efficient Residual Learning with Mixture-of-Experts for Universal Dexterous Grasping","date":"2024-10-03","arxiv_id":"2410.02475","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-expert-estimation-in-hierarchical-mixture","title":"On Expert Estimation in Hierarchical Mixture of Experts: Beyond Softmax Gating Functions","date":"2024-10-03","arxiv_id":"2410.02935","n_code_links":0,"syntology":null},{"paper":"/paper/searching-for-efficient-linear-layers-over-a","slug":"searching-for-efficient-linear-layers-over-a","title":"Searching for Efficient Linear Layers over a Continuous Space of Structured Matrices","date":"2024-10-03","arxiv_id":"2410.02117","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["andpotap/einsum-search"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"ec-dit-scaling-diffusion-transformers-with","title":"EC-DIT: Scaling Diffusion Transformers with Adaptive Expert-Choice Routing","date":"2024-10-02","arxiv_id":"2410.02098","n_code_links":0,"syntology":null},{"paper":null,"slug":"upcycling-instruction-tuning-from-dense-to","title":"Upcycling Instruction Tuning from Dense to Mixture-of-Experts via Parameter Merging","date":"2024-10-02","arxiv_id":"2410.01610","n_code_links":0,"syntology":null},{"paper":null,"slug":"duo-llm-a-framework-for-studying-adaptive","title":"Duo-LLM: A Framework for Studying Adaptive Computation in Large Language Models","date":"2024-10-01","arxiv_id":"2410.10846","n_code_links":0,"syntology":null},{"paper":null,"slug":"idea-an-inverse-domain-expert-adaptation","title":"IDEA: An Inverse Domain Expert Adaptation Based Active DNN IP Protection Method","date":"2024-09-29","arxiv_id":"2410.00059","n_code_links":0,"syntology":null},{"paper":"/paper/clip-moe-towards-building-mixture-of-experts","slug":"clip-moe-towards-building-mixture-of-experts","title":"CLIP-MoE: Towards Building Mixture of Experts for CLIP with Diversified Multiplet Upcycling","date":"2024-09-28","arxiv_id":"2409.19291","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["OpenSparseLLMs/CLIP-MoE"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"boosting-code-switching-asr-with-mixture-of","title":"Boosting Code-Switching ASR with Mixture of Experts Enhanced Speech-Conditioned LLM","date":"2024-09-24","arxiv_id":"2409.15905","n_code_links":0,"syntology":null},{"paper":"/paper/a-gated-residual-kolmogorov-arnold-networks","slug":"a-gated-residual-kolmogorov-arnold-networks","title":"A Gated Residual Kolmogorov-Arnold Networks for Mixtures of Experts","date":"2024-09-23","arxiv_id":"2409.15161","n_code_links":1,"syntology":null},{"paper":"/paper/on-device-collaborative-language-modeling-via","slug":"on-device-collaborative-language-modeling-via","title":"On-Device Collaborative Language Modeling via a Mixture of Generalists and Specialists","date":"2024-09-20","arxiv_id":"2409.13931","n_code_links":1,"syntology":null},{"paper":null,"slug":"retrieval-augmented-test-generation-how-far","title":"Retrieval-Augmented Test Generation: How Far Are We?","date":"2024-09-19","arxiv_id":"2409.12682","n_code_links":0,"syntology":null},{"paper":null,"slug":"grin-gradient-informed-moe","title":"GRIN: GRadient-INformed MoE","date":"2024-09-18","arxiv_id":"2409.12136","n_code_links":0,"syntology":null},{"paper":null,"slug":"mixture-of-diverse-size-experts","title":"Mixture of Diverse Size Experts","date":"2024-09-18","arxiv_id":"2409.12210","n_code_links":0,"syntology":null},{"paper":null,"slug":"da-moe-towards-dynamic-expert-allocation-for","title":"DA-MoE: Towards Dynamic Expert Allocation for Mixture-of-Experts Models","date":"2024-09-10","arxiv_id":"2409.06669","n_code_links":0,"syntology":null},{"paper":null,"slug":"stun-structured-then-unstructured-pruning-for","title":"STUN: Structured-Then-Unstructured Pruning for Scalable MoE Pruning","date":"2024-09-10","arxiv_id":"2409.06211","n_code_links":0,"syntology":null},{"paper":"/paper/alt-moe-multimodal-alignment-via-alternating","slug":"alt-moe-multimodal-alignment-via-alternating","title":"M3-Jepa: Multimodal Alignment via Multi-directional MoE based on the JEPA framework","date":"2024-09-09","arxiv_id":"2409.05929","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["HongyangLL/M3-JEPA"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"chartmoe-mixture-of-expert-connector-for","title":"ChartMoE: Mixture of Expert Connector for Advanced Chart Understanding","date":"2024-09-05","arxiv_id":"2409.03277","n_code_links":0,"syntology":null},{"paper":null,"slug":"interpretable-mixture-of-experts-for-time","title":"Interpretable mixture of experts for time series prediction under recurrent and non-recurrent conditions","date":"2024-09-05","arxiv_id":"2409.03282","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-code-switching-speech-recognition-1","title":"Enhancing Code-Switching Speech Recognition with LID-Based Collaborative Mixture of Experts Model","date":"2024-09-03","arxiv_id":"2409.02050","n_code_links":0,"syntology":null},{"paper":"/paper/olmoe-open-mixture-of-experts-language-models","slug":"olmoe-open-mixture-of-experts-language-models","title":"OLMoE: Open Mixture-of-Experts Language Models","date":"2024-09-03","arxiv_id":"2409.02060","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["allenai/OLMoE"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"beyond-parameter-count-implicit-bias-in-soft","title":"Beyond Parameter Count: Implicit Bias in Soft Mixture of Experts","date":"2024-09-02","arxiv_id":"2409.00879","n_code_links":0,"syntology":null},{"paper":null,"slug":"duplex-a-device-for-large-language-models","title":"Duplex: A Device for Large Language Models with Mixture of Experts, Grouped Query Attention, and Continuous Batching","date":"2024-09-02","arxiv_id":"2409.01141","n_code_links":0,"syntology":null},{"paper":null,"slug":"revisiting-smoe-language-models-by-evaluating","title":"Revisiting SMoE Language Models by Evaluating Inefficiencies with Task Specific Expert Pruning","date":"2024-09-02","arxiv_id":"2409.01483","n_code_links":0,"syntology":null},{"paper":null,"slug":"flexible-and-effective-mixing-of-large","title":"Flexible and Effective Mixing of Large Language Models into a Mixture of Domain Experts","date":"2024-08-30","arxiv_id":"2408.17280","n_code_links":0,"syntology":null},{"paper":null,"slug":"auxiliary-loss-free-load-balancing-strategy","title":"Auxiliary-Loss-Free Load Balancing Strategy for Mixture-of-Experts","date":"2024-08-28","arxiv_id":"2408.15664","n_code_links":0,"syntology":null},{"paper":"/paper/leveraging-open-knowledge-for-advancing-task","slug":"leveraging-open-knowledge-for-advancing-task","title":"Leveraging Open Knowledge for Advancing Task Expertise in Large Language Models","date":"2024-08-28","arxiv_id":"2408.15915","n_code_links":1,"syntology":null},{"paper":null,"slug":"nexus-specialization-meets-adaptability-for","title":"Nexus: Specialization meets Adaptability for Efficiently Training Mixture of Experts","date":"2024-08-28","arxiv_id":"2408.15901","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-survey-of-large-language-models-for-3","title":"A Survey of Large Language Models for European Languages","date":"2024-08-27","arxiv_id":"2408.15040","n_code_links":0,"syntology":null},{"paper":null,"slug":"la-softmoe-clip-for-unified-physical-digital","title":"La-SoftMoE CLIP for Unified Physical-Digital Face Attack Detection","date":"2024-08-23","arxiv_id":"2408.12793","n_code_links":0,"syntology":null},{"paper":"/paper/power-scheduler-a-batch-size-and-token-number","slug":"power-scheduler-a-batch-size-and-token-number","title":"Power Scheduler: A Batch Size and Token Number Agnostic Learning Rate Scheduler","date":"2024-08-23","arxiv_id":"2408.13359","n_code_links":1,"syntology":null},{"paper":null,"slug":"fedmoe-personalized-federated-learning-via","title":"FedMoE: Personalized Federated Learning via Heterogeneous Mixture of Experts","date":"2024-08-21","arxiv_id":"2408.11304","n_code_links":0,"syntology":null},{"paper":"/paper/moe-lpr-multilingual-extension-of-large","slug":"moe-lpr-multilingual-extension-of-large","title":"MoE-LPR: Multilingual Extension of Large Language Models through Mixture-of-Experts with Language Priors Routing","date":"2024-08-21","arxiv_id":"2408.11396","n_code_links":2,"syntology":null},{"paper":null,"slug":"hmoe-heterogeneous-mixture-of-experts-for","title":"HMoE: Heterogeneous Mixture of Experts for Language Modeling","date":"2024-08-20","arxiv_id":"2408.10681","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-unified-framework-for-iris-anti-spoofing","title":"A Unified Framework for Iris Anti-Spoofing: Introducing IrisGeneral Dataset and Masked-MoE Method","date":"2024-08-19","arxiv_id":"2408.09752","n_code_links":0,"syntology":null},{"paper":"/paper/adapmoe-adaptive-sensitivity-based-expert","slug":"adapmoe-adaptive-sensitivity-based-expert","title":"AdapMoE: Adaptive Sensitivity-based Expert Gating and Management for Efficient MoE Inference","date":"2024-08-19","arxiv_id":"2408.10284","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["pku-sec-lab/adapmoe"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/smile-zero-shot-sparse-mixture-of-low-rank","slug":"smile-zero-shot-sparse-mixture-of-low-rank","title":"SMILE: Zero-Shot Sparse Mixture of Low-Rank Experts Construction From Pre-Trained Foundation Models","date":"2024-08-19","arxiv_id":"2408.10174","n_code_links":1,"syntology":null},{"paper":null,"slug":"bam-just-like-that-simple-and-efficient","title":"BAM! Just Like That: Simple and Efficient Parameter Upcycling for Mixture of Experts","date":"2024-08-15","arxiv_id":"2408.08274","n_code_links":0,"syntology":null},{"paper":"/paper/aquilamoe-efficient-training-for-moe-models","slug":"aquilamoe-efficient-training-for-moe-models","title":"AquilaMoE: Efficient Training for MoE Models with Scale-Up and Scale-Out Strategies","date":"2024-08-13","arxiv_id":"2408.06567","n_code_links":1,"syntology":null},{"paper":"/paper/layerwise-recurrent-router-for-mixture-of","slug":"layerwise-recurrent-router-for-mixture-of","title":"Layerwise Recurrent Router for Mixture-of-Experts","date":"2024-08-13","arxiv_id":"2408.06793","n_code_links":1,"syntology":null},{"paper":null,"slug":"home-hierarchy-of-multi-gate-experts-for","title":"HoME: Hierarchy of Multi-Gate Experts for Multi-Task Learning at Kuaishou","date":"2024-08-10","arxiv_id":"2408.05430","n_code_links":0,"syntology":null},{"paper":null,"slug":"ladimo-layer-wise-distillation-inspired","title":"LaDiMo: Layer-wise Distillation Inspired MoEfier","date":"2024-08-08","arxiv_id":"2408.04278","n_code_links":0,"syntology":null},{"paper":null,"slug":"partial-experts-checkpoint-efficient-fault","title":"MoC-System: Efficient Fault Tolerance for Sparse Mixture-of-Experts Model Training","date":"2024-08-08","arxiv_id":"2408.04307","n_code_links":0,"syntology":null},{"paper":"/paper/understanding-the-performance-and-estimating","slug":"understanding-the-performance-and-estimating","title":"Understanding the Performance and Estimating the Cost of LLM Fine-Tuning","date":"2024-08-08","arxiv_id":"2408.04693","n_code_links":1,"syntology":null},{"paper":"/paper/moextend-tuning-new-experts-for-modality-and","slug":"moextend-tuning-new-experts-for-modality-and","title":"MoExtend: Tuning New Experts for Modality and Task Extension","date":"2024-08-07","arxiv_id":"2408.03511","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["zhongshsh/moextend"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"moma-efficient-early-fusion-pre-training-with","title":"MoMa: Efficient Early-Fusion Pre-training with Mixture of Modality-Aware Experts","date":"2024-07-31","arxiv_id":"2407.21770","n_code_links":0,"syntology":null},{"paper":"/paper/mixture-of-modular-experts-distilling","slug":"mixture-of-modular-experts-distilling","title":"Mixture of Modular Experts: Distilling Knowledge from a Multilingual Teacher into Specialized Modular Language Models","date":"2024-07-28","arxiv_id":"2407.19610","n_code_links":1,"syntology":null},{"paper":"/paper/dynamic-language-group-based-moe-enhancing","slug":"dynamic-language-group-based-moe-enhancing","title":"Dynamic Language Group-Based MoE: Enhancing Code-Switching Speech Recognition with Hierarchical Routing","date":"2024-07-26","arxiv_id":"2407.18581","n_code_links":1,"syntology":null},{"paper":null,"slug":"how-lightweight-can-a-vision-transformer-be","title":"How Lightweight Can A Vision Transformer Be","date":"2024-07-25","arxiv_id":"2407.17783","n_code_links":0,"syntology":null},{"paper":null,"slug":"eegmamba-bidirectional-state-space-models","title":"EEGMamba: Bidirectional State Space Model with Mixture of Experts for EEG Multi-task Classification","date":"2024-07-20","arxiv_id":"2407.20254","n_code_links":0,"syntology":null},{"paper":null,"slug":"mixture-of-experts-with-mixture-of-precisions","title":"Mixture of Experts with Mixture of Precisions for Tuning Quality of Service","date":"2024-07-19","arxiv_id":"2407.14417","n_code_links":0,"syntology":null},{"paper":"/paper/scaling-diffusion-transformers-to-16-billion","slug":"scaling-diffusion-transformers-to-16-billion","title":"Scaling Diffusion Transformers to 16 Billion Parameters","date":"2024-07-16","arxiv_id":"2407.11633","n_code_links":1,"syntology":{"ran":12,"of":16,"n_ran_checked":6,"n_instrument":6,"unverified":4,"pointer_only":16,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 4 honoured, 1 violated, 1 with no contract checked; 6 where Syntology's instrument failed) · 4 unverified","official":{"repos":["feizc/dit-moe"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/low-rank-interconnected-adaptation-across","slug":"low-rank-interconnected-adaptation-across","title":"Low-Rank Interconnected Adaptation across Layers","date":"2024-07-13","arxiv_id":"2407.09946","n_code_links":1,"syntology":{"ran":9,"of":13,"n_ran_checked":6,"n_instrument":3,"unverified":4,"pointer_only":13,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","official":{"repos":["yibozhong/lily"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/maskmoe-boosting-token-level-learning-via","slug":"maskmoe-boosting-token-level-learning-via","title":"MaskMoE: Boosting Token-Level Learning via Routing Mask in Mixture-of-Experts","date":"2024-07-13","arxiv_id":"2407.09816","n_code_links":1,"syntology":null},{"paper":null,"slug":"diversifying-the-expert-knowledge-for-task","title":"Diversifying the Expert Knowledge for Task-Agnostic Pruning in Sparse Mixture-of-Experts","date":"2024-07-12","arxiv_id":"2407.09590","n_code_links":0,"syntology":null},{"paper":"/paper/swin-smt-global-sequential-modeling-in-3d","slug":"swin-smt-global-sequential-modeling-in-3d","title":"Swin SMT: Global Sequential Modeling in 3D Medical Image Segmentation","date":"2024-07-10","arxiv_id":"2407.07514","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-simple-architecture-for-enterprise-large","title":"A Simple Architecture for Enterprise Large Language Model Applications based on Role based security and Clearance Levels using Retrieval-Augmented Generation or Mixture of Experts","date":"2024-07-09","arxiv_id":"2407.06718","n_code_links":0,"syntology":null},{"paper":null,"slug":"lazarus-resilient-and-elastic-training-of","title":"Lazarus: Resilient and Elastic Training of Mixture-of-Experts Models with Adaptive Expert Placement","date":"2024-07-05","arxiv_id":"2407.04656","n_code_links":0,"syntology":null},{"paper":"/paper/loco-low-bit-communication-adaptor-for-large","slug":"loco-low-bit-communication-adaptor-for-large","title":"LoCo: Low-Bit Communication Adaptor for Large-scale Model Training","date":"2024-07-05","arxiv_id":"2407.04480","n_code_links":1,"syntology":null},{"paper":"/paper/yourmt3-multi-instrument-music-transcription","slug":"yourmt3-multi-instrument-music-transcription","title":"YourMT3+: Multi-instrument Music Transcription with Enhanced Transformer Architectures and Cross-dataset Stem Augmentation","date":"2024-07-05","arxiv_id":"2407.04822","n_code_links":1,"syntology":null},{"paper":"/paper/mixture-of-a-million-experts","slug":"mixture-of-a-million-experts","title":"Mixture of A Million Experts","date":"2024-07-04","arxiv_id":"2407.04153","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"efficient-empathy-towards-efficient-and","title":"Efficient-Empathy: Towards Efficient and Effective Selection of Empathy Data","date":"2024-07-02","arxiv_id":"2407.01937","n_code_links":0,"syntology":null},{"paper":"/paper/let-the-expert-stick-to-his-last-expert","slug":"let-the-expert-stick-to-his-last-expert","title":"Let the Expert Stick to His Last: Expert-Specialized Fine-Tuning for Sparse Architectural Large Language Models","date":"2024-07-02","arxiv_id":"2407.01906","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":0,"n_instrument":1,"unverified":2,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["deepseek-ai/esft"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/parm-efficient-training-of-large-sparsely","slug":"parm-efficient-training-of-large-sparsely","title":"Parm: Efficient Training of Large Sparsely-Activated Models with Dedicated Schedules","date":"2024-06-30","arxiv_id":"2407.00599","n_code_links":1,"syntology":null},{"paper":null,"slug":"lemoe-advanced-mixture-of-experts-adaptor-for","title":"LEMoE: Advanced Mixture of Experts Adaptor for Lifelong Model Editing of Large Language Models","date":"2024-06-28","arxiv_id":"2406.20030","n_code_links":0,"syntology":null},{"paper":"/paper/solving-token-gradient-conflict-in-mixture-of","slug":"solving-token-gradient-conflict-in-mixture-of","title":"Solving Token Gradient Conflict in Mixture-of-Experts for Large Vision-Language Model","date":"2024-06-28","arxiv_id":"2406.19905","n_code_links":1,"syntology":null},{"paper":"/paper/a-closer-look-into-mixture-of-experts-in","slug":"a-closer-look-into-mixture-of-experts-in","title":"A Closer Look into Mixture-of-Experts in Large Language Models","date":"2024-06-26","arxiv_id":"2406.18219","n_code_links":1,"syntology":null},{"paper":"/paper/a-survey-on-mixture-of-experts","slug":"a-survey-on-mixture-of-experts","title":"A Survey on Mixture of Experts","date":"2024-06-26","arxiv_id":"2407.06204","n_code_links":1,"syntology":null}],"record_sha256":"3fa5d97a2e6b67b784a8b5c4a36871e5ca16ef7fade5bd257d0b798434fb73ca","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}