{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/moe/papers/2","list_of":"/method/moe","method":"MoE","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":2,"pages_in_order":4,"rows_per_page":100,"rows":[101,200],"of":366,"counts":{"archive_papers_tagged":366,"with_a_code_link":142,"where_syntology_ran_a_sample":64,"not_listed_spam_title":0,"listed":366,"listed_where_code_ran":64,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":49,"every_run_a_failure_of_syntologys_instrument":15,"listed_with_a_run_with_no_instrument_failure":49,"listed_every_run_a_failure_of_syntologys_instrument":15,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/moe","prev":"/method/moe","next":"/method/moe/papers/3","papers":[{"paper":null,"slug":"every-flop-counts-scaling-a-300b-mixture-of","title":"Every FLOP Counts: Scaling a 300B Mixture-of-Experts LING LLM without Premium GPUs","date":"2025-03-07","arxiv_id":"2503.05139","n_code_links":0,"syntology":null},{"paper":"/paper/linear-moe-linear-sequence-modeling-meets","slug":"linear-moe-linear-sequence-modeling-meets","title":"Linear-MoE: Linear Sequence Modeling Meets Mixture-of-Experts","date":"2025-03-07","arxiv_id":"2503.05447","n_code_links":1,"syntology":null},{"paper":null,"slug":"continual-pre-training-of-moes-how-robust-is","title":"Continual Pre-training of MoEs: How robust is your router?","date":"2025-03-06","arxiv_id":"2503.05029","n_code_links":0,"syntology":null},{"paper":null,"slug":"speculative-moe-communication-efficient","title":"Speculative MoE: Communication Efficient Parallel MoE Inference with Speculative Token and Expert Pre-scheduling","date":"2025-03-06","arxiv_id":"2503.04398","n_code_links":0,"syntology":null},{"paper":null,"slug":"convergence-rates-for-softmax-gating-mixture","title":"Convergence Rates for Softmax Gating Mixture of Experts","date":"2025-03-05","arxiv_id":"2503.03213","n_code_links":0,"syntology":null},{"paper":"/paper/union-of-experts-adapting-hierarchical","slug":"union-of-experts-adapting-hierarchical","title":"Union of Experts: Adapting Hierarchical Routing to Equivalently Decomposed Transformer","date":"2025-03-04","arxiv_id":"2503.02495","n_code_links":1,"syntology":null},{"paper":null,"slug":"ders-towards-extremely-efficient-upcycled","title":"DeRS: Towards Extremely Efficient Upcycled Mixture-of-Experts Models","date":"2025-03-03","arxiv_id":"2503.01359","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-do-consumers-really-choose-exposing","title":"How Do Consumers Really Choose: Exposing Hidden Preferences with the Mixture of Experts Model","date":"2025-03-03","arxiv_id":"2503.05800","n_code_links":0,"syntology":null},{"paper":null,"slug":"cl-moe-enhancing-multimodal-large-language","title":"CL-MoE: Enhancing Multimodal Large Language Model with Dual Momentum Mixture-of-Experts for Continual Visual Question Answering","date":"2025-03-01","arxiv_id":"2503.00413","n_code_links":0,"syntology":null},{"paper":null,"slug":"cosmoes-compact-sparse-mixture-of-experts","title":"CoSMoEs: Compact Sparse Mixture of Experts","date":"2025-02-28","arxiv_id":"2503.00245","n_code_links":0,"syntology":null},{"paper":"/paper/comet-fine-grained-computation-communication","slug":"comet-fine-grained-computation-communication","title":"Comet: Fine-grained Computation-communication Overlapping for Mixture-of-Experts","date":"2025-02-27","arxiv_id":"2502.19811","n_code_links":4,"syntology":{"ran":9,"of":20,"n_ran_checked":9,"n_instrument":0,"unverified":11,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 11 unverified","official":{"repos":["bytedance/flux"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":11,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"mixture-of-experts-for-recognizing-depression","title":"Mixture of Experts for Recognizing Depression from Interview and Reading Tasks","date":"2025-02-27","arxiv_id":"2502.20213","n_code_links":0,"syntology":null},{"paper":"/paper/r2-t2-re-routing-in-test-time-for-multimodal","slug":"r2-t2-re-routing-in-test-time-for-multimodal","title":"R2-T2: Re-Routing in Test-Time for Multimodal Mixture-of-Experts","date":"2025-02-27","arxiv_id":"2502.20395","n_code_links":1,"syntology":null},{"paper":"/paper/drop-upcycling-training-sparse-mixture-of","slug":"drop-upcycling-training-sparse-mixture-of","title":"Drop-Upcycling: Training Sparse Mixture of Experts with Partial Re-initialization","date":"2025-02-26","arxiv_id":"2502.19261","n_code_links":0,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"bigmac-a-communication-efficient-mixture-of","title":"BigMac: A Communication-Efficient Mixture-of-Experts Model Structure for Fast Training and Inference","date":"2025-02-24","arxiv_id":"2502.16927","n_code_links":0,"syntology":null},{"paper":"/paper/delta-decompression-for-moe-based-llms","slug":"delta-decompression-for-moe-based-llms","title":"Delta Decompression for MoE-based LLMs Compression","date":"2025-02-24","arxiv_id":"2502.17298","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":0,"n_instrument":2,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["lliai/d2moe"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"evaluating-expert-contributions-in-a-moe-llm","title":"Evaluating Expert Contributions in a MoE LLM for Quiz-Based Tasks","date":"2025-02-24","arxiv_id":"2502.17187","n_code_links":0,"syntology":null},{"paper":"/paper/make-lora-great-again-boosting-lora-with","slug":"make-lora-great-again-boosting-lora-with","title":"Make LoRA Great Again: Boosting LoRA with Adaptive Singular Values and Mixture-of-Experts Optimization Alignment","date":"2025-02-24","arxiv_id":"2502.16894","n_code_links":1,"syntology":{"ran":6,"of":8,"n_ran_checked":3,"n_instrument":3,"unverified":2,"pointer_only":0,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["facico/goat-peft"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/the-empirical-impact-of-reducing-symmetries","slug":"the-empirical-impact-of-reducing-symmetries","title":"The Empirical Impact of Reducing Symmetries on the Performance of Deep Ensembles and MoE","date":"2025-02-24","arxiv_id":"2502.17391","n_code_links":0,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":7,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 6 with no instrument failure: 3 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/binary-integer-programming-based-algorithm","slug":"binary-integer-programming-based-algorithm","title":"Binary-Integer-Programming Based Algorithm for Expert Load Balancing in Mixture-of-Experts Models","date":"2025-02-21","arxiv_id":"2502.15451","n_code_links":1,"syntology":null},{"paper":"/paper/tight-clusters-make-specialized-experts","slug":"tight-clusters-make-specialized-experts","title":"Tight Clusters Make Specialized Experts","date":"2025-02-21","arxiv_id":"2502.15315","n_code_links":1,"syntology":null},{"paper":null,"slug":"dsmoe-matrix-partitioned-experts-with-dynamic","title":"DSMoE: Matrix-Partitioned Experts with Dynamic Routing for Computation-Efficient Dense LLMs","date":"2025-02-18","arxiv_id":"2502.12455","n_code_links":0,"syntology":null},{"paper":null,"slug":"every-expert-matters-towards-effective","title":"Every Expert Matters: Towards Effective Knowledge Distillation for Mixture-of-Experts Language Models","date":"2025-02-18","arxiv_id":"2502.12947","n_code_links":0,"syntology":null},{"paper":"/paper/accurate-expert-predictions-in-moe-inference","slug":"accurate-expert-predictions-in-moe-inference","title":"Fate: Fast Edge Inference of Mixture-of-Experts Models via Cross-Layer Gate","date":"2025-02-17","arxiv_id":"2502.12224","n_code_links":1,"syntology":null},{"paper":null,"slug":"semantic-specialization-in-moe-appears-with","title":"Probing Semantic Routing in Large Mixture-of-Expert Models","date":"2025-02-15","arxiv_id":"2502.10928","n_code_links":0,"syntology":null},{"paper":null,"slug":"mohave-mixture-of-hierarchical-audio-visual","title":"MoHAVE: Mixture of Hierarchical Audio-Visual Experts for Robust Speech Recognition","date":"2025-02-11","arxiv_id":"2502.10447","n_code_links":0,"syntology":null},{"paper":"/paper/training-sparse-mixture-of-experts-text","slug":"training-sparse-mixture-of-experts-text","title":"Training Sparse Mixture Of Experts Text Embedding Models","date":"2025-02-11","arxiv_id":"2502.07972","n_code_links":1,"syntology":{"ran":9,"of":9,"n_ran_checked":9,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["nomic-ai/contrastors"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"moetuner-optimized-mixture-of-expert-serving","title":"MoETuner: Optimized Mixture of Expert Serving with Balanced Expert Placement and Token Routing","date":"2025-02-10","arxiv_id":"2502.06643","n_code_links":0,"syntology":null},{"paper":"/paper/klotski-efficient-mixture-of-expert-inference","slug":"klotski-efficient-mixture-of-expert-inference","title":"Klotski: Efficient Mixture-of-Expert Inference via Expert-Aware Multi-Batch Pipeline","date":"2025-02-09","arxiv_id":"2502.06888","n_code_links":1,"syntology":null},{"paper":null,"slug":"fmoe-fine-grained-expert-offloading-for-large","title":"fMoE: Fine-Grained Expert Offloading for Large Mixture-of-Experts Serving","date":"2025-02-07","arxiv_id":"2502.05370","n_code_links":0,"syntology":null},{"paper":null,"slug":"joint-moe-scaling-laws-mixture-of-experts-can","title":"Joint MoE Scaling Laws: Mixture of Experts Can Be Memory Efficient","date":"2025-02-07","arxiv_id":"2502.05172","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-foundational-models-for-dynamical","title":"Towards Foundational Models for Dynamical System Reconstruction: Hierarchical Meta-Learning via Mixture of Experts","date":"2025-02-07","arxiv_id":"2502.05335","n_code_links":0,"syntology":null},{"paper":"/paper/cmoe-fast-carving-of-mixture-of-experts-for","slug":"cmoe-fast-carving-of-mixture-of-experts-for","title":"CMoE: Fast Carving of Mixture-of-Experts for Efficient LLM Inference","date":"2025-02-06","arxiv_id":"2502.04416","n_code_links":1,"syntology":null},{"paper":null,"slug":"mixture-of-neural-operator-experts-for","title":"Mixture of neural operator experts for learning boundary conditions and model selection","date":"2025-02-06","arxiv_id":"2502.04562","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimizing-robustness-and-accuracy-in-mixture","title":"Optimizing Robustness and Accuracy in Mixture of Experts: A Dual-Model Approach","date":"2025-02-05","arxiv_id":"2502.06832","n_code_links":0,"syntology":null},{"paper":null,"slug":"clip-up-a-simple-and-efficient-mixture-of","title":"CLIP-UP: A Simple and Efficient Mixture-of-Experts CLIP Training Recipe with Sparse Upcycling","date":"2025-02-03","arxiv_id":"2502.00965","n_code_links":0,"syntology":null},{"paper":null,"slug":"mergeme-model-merging-techniques-for","title":"MergeME: Model Merging Techniques for Homogeneous and Heterogeneous MoEs","date":"2025-02-03","arxiv_id":"2502.00997","n_code_links":0,"syntology":null},{"paper":null,"slug":"omni-mol-exploring-universal-convergent-space","title":"Omni-Mol: Exploring Universal Convergent Space for Omni-Molecular Tasks","date":"2025-02-03","arxiv_id":"2502.01074","n_code_links":0,"syntology":null},{"paper":null,"slug":"static-batching-of-irregular-workloads-on","title":"Static Batching of Irregular Workloads on GPUs: Framework and Application to Efficient MoE Model Inference","date":"2025-01-27","arxiv_id":"2501.16103","n_code_links":0,"syntology":null},{"paper":null,"slug":"each-rank-could-be-an-expert-single-ranked","title":"Each Rank Could be an Expert: Single-Ranked Mixture of Experts LoRA for Multi-Task Learning","date":"2025-01-25","arxiv_id":"2501.15103","n_code_links":0,"syntology":null},{"paper":"/paper/hierarchical-time-aware-mixture-of-experts","slug":"hierarchical-time-aware-mixture-of-experts","title":"Hierarchical Time-Aware Mixture of Experts for Multi-Modal Sequential Recommendation","date":"2025-01-24","arxiv_id":"2501.14269","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["SStarCCat/HM4SR"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"mean-field-limit-from-general-mixtures-of","title":"Mean-field limit from general mixtures of experts to quantum neural networks","date":"2025-01-24","arxiv_id":"2501.14660","n_code_links":0,"syntology":null},{"paper":null,"slug":"autonomy-of-experts-models","title":"Autonomy-of-Experts Models","date":"2025-01-22","arxiv_id":"2501.13074","n_code_links":0,"syntology":null},{"paper":null,"slug":"blr-moe-boosted-language-routing-mixture-of","title":"BLR-MoE: Boosted Language-Routing Mixture of Experts for Domain-Robust Multilingual E2E ASR","date":"2025-01-22","arxiv_id":"2501.12602","n_code_links":0,"syntology":null},{"paper":null,"slug":"demons-in-the-detail-on-implementing-load","title":"Demons in the Detail: On Implementing Load Balancing Loss for Training Specialized Mixture-of-Expert Models","date":"2025-01-21","arxiv_id":"2501.11873","n_code_links":0,"syntology":null},{"paper":null,"slug":"scfcrc-simultaneously-counteract-feature","title":"SCFCRC: Simultaneously Counteract Feature Camouflage and Relation Camouflage for Fraud Detection","date":"2025-01-21","arxiv_id":"2501.12430","n_code_links":0,"syntology":null},{"paper":null,"slug":"fsmoe-a-flexible-and-scalable-training-system","title":"FSMoE: A Flexible and Scalable Training System for Sparse Mixture-of-Experts Models","date":"2025-01-18","arxiv_id":"2501.10714","n_code_links":0,"syntology":null},{"paper":null,"slug":"omoe-diversifying-mixture-of-low-rank","title":"OMoE: Diversifying Mixture of Low-Rank Adaptation by Orthogonal Finetuning","date":"2025-01-17","arxiv_id":"2501.10062","n_code_links":0,"syntology":null},{"paper":null,"slug":"llm-based-routing-in-mixture-of-experts-a","title":"LLM-Based Routing in Mixture of Experts: A Novel Framework for Trading","date":"2025-01-16","arxiv_id":"2501.09636","n_code_links":0,"syntology":null},{"paper":null,"slug":"moe-2-optimizing-collaborative-inference-for","title":"MoE$^2$: Optimizing Collaborative Inference for Edge Large Language Models","date":"2025-01-16","arxiv_id":"2501.09410","n_code_links":0,"syntology":null},{"paper":null,"slug":"graphmoe-amplifying-cognitive-depth-of","title":"GRAPHMOE: Amplifying Cognitive Depth of Mixture-of-Experts Network via Introducing Self-Rethinking Mechanism","date":"2025-01-14","arxiv_id":"2501.07890","n_code_links":0,"syntology":null},{"paper":"/paper/minimax-01-scaling-foundation-models-with","slug":"minimax-01-scaling-foundation-models-with","title":"MiniMax-01: Scaling Foundation Models with Lightning Attention","date":"2025-01-14","arxiv_id":"2501.08313","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"a-comprehensive-evaluation-of-large-language-5","title":"A Comprehensive Evaluation of Large Language Models on Mental Illnesses in Arabic Context","date":"2025-01-12","arxiv_id":"2501.06859","n_code_links":0,"syntology":null},{"paper":"/paper/transforming-vision-transformer-towards","slug":"transforming-vision-transformer-towards","title":"Transforming Vision Transformer: Towards Efficient Multi-Task Asynchronous Learning","date":"2025-01-12","arxiv_id":"2501.06884","n_code_links":1,"syntology":{"ran":3,"of":7,"n_ran_checked":1,"n_instrument":2,"unverified":4,"pointer_only":7,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["yewen1486/emtal"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/tamer-a-test-time-adaptive-moe-driven","slug":"tamer-a-test-time-adaptive-moe-driven","title":"TAMER: A Test-Time Adaptive MoE-Driven Framework for EHR Representation Learning","date":"2025-01-10","arxiv_id":"2501.05661","n_code_links":1,"syntology":null},{"paper":null,"slug":"optimizing-distributed-deployment-of-mixture","title":"Optimizing Distributed Deployment of Mixture-of-Experts Model Inference in Serverless Computing","date":"2025-01-09","arxiv_id":"2501.05313","n_code_links":0,"syntology":null},{"paper":"/paper/limoe-mixture-of-lidar-representation","slug":"limoe-mixture-of-lidar-representation","title":"LiMoE: Mixture of LiDAR Representation Learners from Automotive Scenes","date":"2025-01-07","arxiv_id":"2501.04004","n_code_links":1,"syntology":null},{"paper":null,"slug":"mfabric-an-efficient-and-scalable-fabric-for","title":"mFabric: An Efficient and Scalable Fabric for Mixture-of-Experts Training","date":"2025-01-07","arxiv_id":"2501.03905","n_code_links":0,"syntology":null},{"paper":null,"slug":"fresh-cl-feature-realignment-through-experts","title":"Fresh-CL: Feature Realignment through Experts on Hypersphere in Continual Learning","date":"2025-01-04","arxiv_id":"2501.02198","n_code_links":0,"syntology":null},{"paper":"/paper/sm3det-a-unified-model-for-multi-modal-remote","slug":"sm3det-a-unified-model-for-multi-modal-remote","title":"SM3Det: A Unified Model for Multi-Modal Remote Sensing Object Detection","date":"2024-12-30","arxiv_id":"2412.20665","n_code_links":1,"syntology":null},{"paper":null,"slug":"multimodal-variational-autoencoder-a","title":"Multimodal Variational Autoencoder: a Barycentric View","date":"2024-12-29","arxiv_id":"2412.20487","n_code_links":0,"syntology":null},{"paper":"/paper/askchart-universal-chart-understanding","slug":"askchart-universal-chart-understanding","title":"AskChart: Universal Chart Understanding through Textual Enhancement","date":"2024-12-26","arxiv_id":"2412.19146","n_code_links":1,"syntology":null},{"paper":"/paper/big-moe-bypass-isolated-gating-moe-for","slug":"big-moe-bypass-isolated-gating-moe-for","title":"BIG-MoE: Bypass Isolated Gating MoE for Generalized Multimodal Face Anti-Spoofing","date":"2024-12-24","arxiv_id":"2412.18065","n_code_links":1,"syntology":null},{"paper":null,"slug":"ume-upcycling-mixture-of-experts-for-scalable","title":"UME: Upcycling Mixture-of-Experts for Scalable and Efficient Automatic Speech Recognition","date":"2024-12-23","arxiv_id":"2412.17507","n_code_links":0,"syntology":null},{"paper":null,"slug":"part-of-speech-sensitivity-of-routers-in","title":"Part-Of-Speech Sensitivity of Routers in Mixture of Experts Models","date":"2024-12-22","arxiv_id":"2412.16971","n_code_links":0,"syntology":null},{"paper":null,"slug":"theory-of-mixture-of-experts-for-mobile-edge","title":"Theory of Mixture-of-Experts for Mobile Edge Computing","date":"2024-12-20","arxiv_id":"2412.15690","n_code_links":0,"syntology":null},{"paper":"/paper/remoe-fully-differentiable-mixture-of-experts","slug":"remoe-fully-differentiable-mixture-of-experts","title":"ReMoE: Fully Differentiable Mixture-of-Experts with ReLU Routing","date":"2024-12-19","arxiv_id":"2412.14711","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["thu-ml/remoe"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-survey-on-inference-optimization-techniques","slug":"a-survey-on-inference-optimization-techniques","title":"A Survey on Inference Optimization Techniques for Mixture of Experts Models","date":"2024-12-18","arxiv_id":"2412.14219","n_code_links":1,"syntology":null},{"paper":null,"slug":"graphlora-empowering-llms-fine-tuning-via","title":"GraphLoRA: Empowering LLMs Fine-Tuning via Graph Collaboration of MoE","date":"2024-12-18","arxiv_id":"2412.16216","n_code_links":0,"syntology":null},{"paper":"/paper/seke-specialised-experts-for-keyword","slug":"seke-specialised-experts-for-keyword","title":"SEKE: Specialised Experts for Keyword Extraction","date":"2024-12-18","arxiv_id":"2412.14087","n_code_links":1,"syntology":null},{"paper":"/paper/daop-data-aware-offloading-and-predictive-pre","slug":"daop-data-aware-offloading-and-predictive-pre","title":"DAOP: Data-Aware Offloading and Predictive Pre-Calculation for Efficient MoE Inference","date":"2024-12-16","arxiv_id":"2501.10375","n_code_links":1,"syntology":null},{"paper":null,"slug":"enhancing-healthcare-recommendation-systems","title":"Enhancing Healthcare Recommendation Systems with a Multimodal LLMs-based MOE Architecture","date":"2024-12-16","arxiv_id":"2412.11557","n_code_links":0,"syntology":null},{"paper":null,"slug":"investigating-mixture-of-experts-in-dense","title":"Investigating Mixture of Experts in Dense Retrieval","date":"2024-12-16","arxiv_id":"2412.11864","n_code_links":0,"syntology":null},{"paper":"/paper/llama-3-meets-moe-efficient-upcycling","slug":"llama-3-meets-moe-efficient-upcycling","title":"Llama 3 Meets MoE: Efficient Upcycling","date":"2024-12-13","arxiv_id":"2412.09952","n_code_links":1,"syntology":null},{"paper":null,"slug":"mosld-an-extremely-parameter-efficient","title":"MoSLD: An Extremely Parameter-Efficient Mixture-of-Shared LoRAs for Multi-Task Learning","date":"2024-12-12","arxiv_id":"2412.08946","n_code_links":0,"syntology":null},{"paper":"/paper/towards-a-multimodal-large-language-model","slug":"towards-a-multimodal-large-language-model","title":"Towards a Multimodal Large Language Model with Pixel-Level Insight for Biomedicine","date":"2024-12-12","arxiv_id":"2412.09278","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["shawnhuang497/medplib"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"moe-cap-cost-accuracy-performance","title":"MoE-CAP: Benchmarking Cost, Accuracy and Performance of Sparse Mixture-of-Experts Systems","date":"2024-12-10","arxiv_id":"2412.07067","n_code_links":0,"syntology":null},{"paper":"/paper/post-training-statistical-calibration-for","slug":"post-training-statistical-calibration-for","title":"Post-Training Statistical Calibration for Higher Activation Sparsity","date":"2024-12-10","arxiv_id":"2412.07174","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["intellabs/scap"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/object-detection-using-event-camera-a-moe","slug":"object-detection-using-event-camera-a-moe","title":"Object Detection using Event Camera: A MoE Heat Conduction based Detector and A New Benchmark Dataset","date":"2024-12-09","arxiv_id":"2412.06647","n_code_links":1,"syntology":null},{"paper":null,"slug":"customize-segment-anything-model-for-multi","title":"Customize Segment Anything Model for Multi-Modal Semantic Segmentation with Mixture of LoRA Experts","date":"2024-12-05","arxiv_id":"2412.04220","n_code_links":0,"syntology":null},{"paper":null,"slug":"convolutional-neural-networks-and-mixture-of","title":"Convolutional Neural Networks and Mixture of Experts for Intrusion Detection in 5G Networks and beyond","date":"2024-12-04","arxiv_id":"2412.03483","n_code_links":0,"syntology":null},{"paper":null,"slug":"ca-moe-channel-adapted-moe-for-incremental","title":"CA-MoE: Channel-Adapted MoE for Incremental Weather Forecasting","date":"2024-12-03","arxiv_id":"2412.02503","n_code_links":0,"syntology":null},{"paper":null,"slug":"mixture-of-cache-conditional-experts-for","title":"Mixture of Cache-Conditional Experts for Efficient Mobile Device Inference","date":"2024-11-27","arxiv_id":"2412.00099","n_code_links":0,"syntology":null},{"paper":null,"slug":"mixture-of-experts-in-image-classification","title":"Mixture of Experts in Image Classification: What's the Sweet Spot?","date":"2024-11-27","arxiv_id":"2411.18322","n_code_links":0,"syntology":null},{"paper":null,"slug":"uoe-unlearning-one-expert-is-enough-for","title":"UOE: Unlearning One Expert Is Enough For Mixture-of-experts LLMS","date":"2024-11-27","arxiv_id":"2411.18797","n_code_links":0,"syntology":null},{"paper":"/paper/condense-don-t-just-prune-enhancing","slug":"condense-don-t-just-prune-enhancing","title":"Condense, Don't Just Prune: Enhancing Efficiency and Performance in MoE Layer Pruning","date":"2024-11-26","arxiv_id":"2412.00069","n_code_links":1,"syntology":null},{"paper":null,"slug":"mh-moe-multi-head-mixture-of-experts","title":"MH-MoE: Multi-Head Mixture-of-Experts","date":"2024-11-25","arxiv_id":"2411.16205","n_code_links":0,"syntology":null},{"paper":"/paper/llama-moe-v2-exploring-sparsity-of-llama-from","slug":"llama-moe-v2-exploring-sparsity-of-llama-from","title":"LLaMA-MoE v2: Exploring Sparsity of LLaMA from Perspective of Mixture-of-Experts with Post-Training","date":"2024-11-24","arxiv_id":"2411.15708","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["opensparsellms/llama-moe-v2"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"merlot-a-distilled-llm-based-mixture-of","title":"MERLOT: A Distilled LLM-based Mixture-of-Experts Framework for Scalable Encrypted Traffic Classification","date":"2024-11-20","arxiv_id":"2411.13004","n_code_links":0,"syntology":null},{"paper":null,"slug":"moe-lightning-high-throughput-moe-inference","title":"MoE-Lightning: High-Throughput MoE Inference on Memory-constrained GPUs","date":"2024-11-18","arxiv_id":"2411.11217","n_code_links":0,"syntology":null},{"paper":null,"slug":"lynx-enabling-efficient-moe-inference-through","title":"Lynx: Enabling Efficient MoE Inference through Dynamic Batch-Aware Expert Selection","date":"2024-11-13","arxiv_id":"2411.08982","n_code_links":0,"syntology":null},{"paper":null,"slug":"perft-parameter-efficient-routed-fine-tuning","title":"PERFT: Parameter-Efficient Routed Fine-Tuning for Mixture-of-Expert Model","date":"2024-11-12","arxiv_id":"2411.08212","n_code_links":0,"syntology":null},{"paper":null,"slug":"wdmoe-wireless-distributed-mixture-of-experts","title":"WDMoE: Wireless Distributed Mixture of Experts for Large Language Models","date":"2024-11-11","arxiv_id":"2411.06681","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-mixtures-of-experts-with-em","title":"Learning Mixtures of Experts with EM","date":"2024-11-09","arxiv_id":"2411.06056","n_code_links":0,"syntology":null},{"paper":null,"slug":"neko-toward-post-recognition-generative","title":"NeKo: Toward Post Recognition Generative Correction Large Language Models with Task-Oriented Experts","date":"2024-11-08","arxiv_id":"2411.05945","n_code_links":0,"syntology":null},{"paper":null,"slug":"fedmoe-da-federated-mixture-of-experts-via","title":"FedMoE-DA: Federated Mixture of Experts via Domain Aware Fine-grained Aggregation","date":"2024-11-04","arxiv_id":"2411.02115","n_code_links":0,"syntology":null},{"paper":null,"slug":"facet-aware-multi-head-mixture-of-experts","title":"Facet-Aware Multi-Head Mixture-of-Experts Model for Sequential Recommendation","date":"2024-11-03","arxiv_id":"2411.01457","n_code_links":0,"syntology":null},{"paper":null,"slug":"hobbit-a-mixed-precision-expert-offloading","title":"HOBBIT: A Mixed Precision Expert Offloading System for Fast MoE Inference","date":"2024-11-03","arxiv_id":"2411.01433","n_code_links":0,"syntology":null},{"paper":null,"slug":"rs-moe-mixture-of-experts-for-remote-sensing","title":"RS-MoE: Mixture of Experts for Remote Sensing Image Captioning and Visual Question Answering","date":"2024-11-03","arxiv_id":"2411.01595","n_code_links":0,"syntology":null},{"paper":null,"slug":"pmol-parameter-efficient-moe-for-preference","title":"PMoL: Parameter Efficient MoE for Preference Mixing of LLM Alignment","date":"2024-11-02","arxiv_id":"2411.01245","n_code_links":0,"syntology":null}],"record_sha256":"83758b5f3108e59da1bed83cd25a7e056eefef0f4364f90db0b602fbb297b1d2","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}