{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/moe/papers/4","list_of":"/method/moe","method":"MoE","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":4,"pages_in_order":4,"rows_per_page":100,"rows":[301,366],"of":366,"counts":{"archive_papers_tagged":366,"with_a_code_link":142,"where_syntology_ran_a_sample":64,"not_listed_spam_title":0,"listed":366,"listed_where_code_ran":64,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":49,"every_run_a_failure_of_syntologys_instrument":15,"listed_with_a_run_with_no_instrument_failure":49,"listed_every_run_a_failure_of_syntologys_instrument":15,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/moe","prev":"/method/moe/papers/3","next":null,"papers":[{"paper":null,"slug":"mixture-of-experts-in-a-mixture-of-rl","title":"Mixture of Experts in a Mixture of RL settings","date":"2024-06-26","arxiv_id":"2406.18420","n_code_links":0,"syntology":null},{"paper":null,"slug":"sc-moe-switch-conformer-mixture-of-experts","title":"SC-MoE: Switch Conformer Mixture of Experts for Unified Streaming and Non-streaming Code-Switching ASR","date":"2024-06-26","arxiv_id":"2406.18021","n_code_links":0,"syntology":null},{"paper":null,"slug":"moe-ct-a-novel-approach-for-large-language","title":"MoE-CT: A Novel Approach For Large Language Models Training With Resistance To Catastrophic Forgetting","date":"2024-06-25","arxiv_id":"2407.00875","n_code_links":0,"syntology":null},{"paper":"/paper/llama-moe-building-mixture-of-experts-from","slug":"llama-moe-building-mixture-of-experts-from","title":"LLaMA-MoE: Building Mixture-of-Experts from LLaMA with Continual Pre-training","date":"2024-06-24","arxiv_id":"2406.16554","n_code_links":2,"syntology":null},{"paper":null,"slug":"theory-on-mixture-of-experts-in-continual","title":"Theory on Mixture-of-Experts in Continual Learning","date":"2024-06-24","arxiv_id":"2406.16437","n_code_links":0,"syntology":null},{"paper":"/paper/adamoe-token-adaptive-routing-with-null","slug":"adamoe-token-adaptive-routing-with-null","title":"AdaMoE: Token-Adaptive Routing with Null Experts for Mixture-of-Experts Language Models","date":"2024-06-19","arxiv_id":"2406.13233","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":2,"n_instrument":1,"unverified":2,"pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["cengzihao/adamoe"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/gw-moe-resolving-uncertainty-in-moe-router","slug":"gw-moe-resolving-uncertainty-in-moe-router","title":"GW-MoE: Resolving Uncertainty in MoE Router with Global Workspace Theory","date":"2024-06-18","arxiv_id":"2406.12375","n_code_links":1,"syntology":null},{"paper":null,"slug":"variational-distillation-of-diffusion","title":"Variational Distillation of Diffusion Policies into Mixture of Experts","date":"2024-06-18","arxiv_id":"2406.12538","n_code_links":0,"syntology":null},{"paper":"/paper/dynamic-data-mixing-maximizes-instruction","slug":"dynamic-data-mixing-maximizes-instruction","title":"Dynamic Data Mixing Maximizes Instruction Tuning for Mixture-of-Experts","date":"2024-06-17","arxiv_id":"2406.11256","n_code_links":1,"syntology":{"ran":8,"of":8,"n_ran_checked":5,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["spico197/moe-sft"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/texttt-moe-rbench-towards-building-reliable","slug":"texttt-moe-rbench-towards-building-reliable","title":"$\\texttt{MoE-RBench}$: Towards Building Reliable Language Models with Sparse Mixture-of-Experts","date":"2024-06-17","arxiv_id":"2406.11353","n_code_links":1,"syntology":null},{"paper":"/paper/towards-efficient-pareto-set-approximation","slug":"towards-efficient-pareto-set-approximation","title":"Towards Efficient Pareto Set Approximation via Mixture of Experts Based Model Fusion","date":"2024-06-14","arxiv_id":"2406.09770","n_code_links":1,"syntology":null},{"paper":"/paper/examining-post-training-quantization-for","slug":"examining-post-training-quantization-for","title":"Examining Post-Training Quantization for Mixture-of-Experts: A Benchmark","date":"2024-06-12","arxiv_id":"2406.08155","n_code_links":1,"syntology":{"ran":1,"of":6,"n_ran_checked":0,"n_instrument":1,"unverified":5,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":{"repos":["unites-lab/moe-quantization"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/moe-jetpack-from-dense-checkpoints-to","slug":"moe-jetpack-from-dense-checkpoints-to","title":"MoE Jetpack: From Dense Checkpoints to Adaptive Mixture of Experts for Vision Tasks","date":"2024-06-07","arxiv_id":"2406.04801","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":5,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["adlith/moe-jetpack"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"filtered-not-mixed-stochastic-filtering-based","title":"Filtered not Mixed: Stochastic Filtering-Based Online Gating for Mixture of Large Language Models","date":"2024-06-05","arxiv_id":"2406.02969","n_code_links":0,"syntology":null},{"paper":null,"slug":"style-mixture-of-experts-for-expressive-text","title":"Style Mixture of Experts for Expressive Text-To-Speech Synthesis","date":"2024-06-05","arxiv_id":"2406.03637","n_code_links":0,"syntology":null},{"paper":"/paper/demystifying-the-compression-of-mixture-of","slug":"demystifying-the-compression-of-mixture-of","title":"Demystifying the Compression of Mixture-of-Experts Through a Unified Framework","date":"2024-06-04","arxiv_id":"2406.02500","n_code_links":1,"syntology":{"ran":5,"of":11,"n_ran_checked":5,"n_instrument":0,"unverified":6,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","official":{"repos":["daizedong/unified-moe-compression"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/leveraging-visual-tokens-for-extended-text","slug":"leveraging-visual-tokens-for-extended-text","title":"Leveraging Visual Tokens for Extended Text Contexts in Multi-Modal Learning","date":"2024-06-04","arxiv_id":"2406.02547","n_code_links":1,"syntology":{"ran":9,"of":14,"n_ran_checked":9,"n_instrument":0,"unverified":5,"pointer_only":14,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 2 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["showlab/VisInContext"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/parrot-multilingual-visual-instruction-tuning","slug":"parrot-multilingual-visual-instruction-tuning","title":"Parrot: Multilingual Visual Instruction Tuning","date":"2024-06-04","arxiv_id":"2406.02539","n_code_links":2,"syntology":{"ran":3,"of":7,"n_ran_checked":0,"n_instrument":3,"unverified":4,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","official":{"repos":["aidc-ai/parrot"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/demystifying-platform-requirements-for","slug":"demystifying-platform-requirements-for","title":"Demystifying AI Platform Design for Distributed Inference of Next-Generation LLM models","date":"2024-06-03","arxiv_id":"2406.01698","n_code_links":1,"syntology":null},{"paper":"/paper/skywork-moe-a-deep-dive-into-training","slug":"skywork-moe-a-deep-dive-into-training","title":"Skywork-MoE: A Deep Dive into Training Techniques for Mixture-of-Experts Language Models","date":"2024-06-03","arxiv_id":"2406.06563","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"optimizing-6g-integrated-sensing-and","title":"Optimizing 6G Integrated Sensing and Communications (ISAC) via Expert Networks","date":"2024-06-01","arxiv_id":"2406.00408","n_code_links":0,"syntology":null},{"paper":null,"slug":"memoe-enhancing-model-editing-with-mixture-of","title":"MEMoE: Enhancing Model Editing with Mixture of Experts Adaptors","date":"2024-05-29","arxiv_id":"2405.19086","n_code_links":0,"syntology":null},{"paper":null,"slug":"monde-mixture-of-near-data-experts-for-large","title":"MoNDE: Mixture of Near-Data Experts for Large-Scale Sparse Models","date":"2024-05-29","arxiv_id":"2405.18832","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-provably-effective-method-for-pruning","title":"A Provably Effective Method for Pruning Experts in Fine-tuned Sparse Mixture-of-Experts","date":"2024-05-26","arxiv_id":"2405.16646","n_code_links":0,"syntology":null},{"paper":null,"slug":"expert-token-resonance-redefining-moe-routing","title":"Expert-Token Resonance: Redefining MoE Routing through Affinity-Driven Active Selection","date":"2024-05-24","arxiv_id":"2406.00023","n_code_links":0,"syntology":null},{"paper":"/paper/revisiting-moe-and-dense-speed-accuracy","slug":"revisiting-moe-and-dense-speed-accuracy","title":"Revisiting MoE and Dense Speed-Accuracy Comparisons for LLM Training","date":"2024-05-23","arxiv_id":"2405.15052","n_code_links":1,"syntology":null},{"paper":null,"slug":"statistical-advantages-of-perturbing-cosine","title":"Statistical Advantages of Perturbing Cosine Router in Mixture of Experts","date":"2024-05-23","arxiv_id":"2405.14131","n_code_links":0,"syntology":null},{"paper":"/paper/unchosen-experts-can-contribute-too","slug":"unchosen-experts-can-contribute-too","title":"Unchosen Experts Can Contribute Too: Unleashing MoE Models' Power by Self-Contrast","date":"2024-05-23","arxiv_id":"2405.14507","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["davidfanzz/scmoe"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/meteora-multiple-tasks-embedded-lora-for","slug":"meteora-multiple-tasks-embedded-lora-for","title":"MeteoRA: Multiple-tasks Embedded LoRA for Large Language Models","date":"2024-05-19","arxiv_id":"2405.13053","n_code_links":1,"syntology":{"ran":8,"of":10,"n_ran_checked":5,"n_instrument":3,"unverified":2,"pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 3 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["paragonlight/meteor-of-lora"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/uni-moe-scaling-unified-multimodal-llms-with","slug":"uni-moe-scaling-unified-multimodal-llms-with","title":"Uni-MoE: Scaling Unified Multimodal LLMs with Mixture of Experts","date":"2024-05-18","arxiv_id":"2405.11273","n_code_links":1,"syntology":{"ran":6,"of":8,"n_ran_checked":3,"n_instrument":3,"unverified":2,"pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["hitsz-tmg/umoe-scaling-unified-multimodal-llms"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/harnessing-hierarchical-label-distribution","slug":"harnessing-hierarchical-label-distribution","title":"Harnessing Hierarchical Label Distribution Variations in Test Agnostic Long-tail Recognition","date":"2024-05-13","arxiv_id":"2405.07780","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["scongl/dirmixe"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-mixture-of-experts-approach-to-3d-human","slug":"a-mixture-of-experts-approach-to-3d-human","title":"A Mixture of Experts Approach to 3D Human Motion Prediction","date":"2024-05-09","arxiv_id":"2405.06088","n_code_links":1,"syntology":null},{"paper":"/paper/cumo-scaling-multimodal-llm-with-co-upcycled","slug":"cumo-scaling-multimodal-llm-with-co-upcycled","title":"CuMo: Scaling Multimodal LLM with Co-Upcycled Mixture-of-Experts","date":"2024-05-09","arxiv_id":"2405.05949","n_code_links":1,"syntology":{"ran":11,"of":12,"n_ran_checked":6,"n_instrument":5,"unverified":1,"pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","official":{"repos":["shi-labs/cumo"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/ewmoe-an-effective-model-for-global-weather","slug":"ewmoe-an-effective-model-for-global-weather","title":"EWMoE: An effective model for global weather forecasting with mixture-of-experts","date":"2024-05-09","arxiv_id":"2405.06004","n_code_links":1,"syntology":null},{"paper":null,"slug":"lory-fully-differentiable-mixture-of-experts","title":"Lory: Fully Differentiable Mixture-of-Experts for Autoregressive Language Model Pre-training","date":"2024-05-06","arxiv_id":"2405.03133","n_code_links":0,"syntology":null},{"paper":null,"slug":"wdmoe-wireless-distributed-large-language","title":"WDMoE: Wireless Distributed Large Language Models with Mixture of Experts","date":"2024-05-06","arxiv_id":"2405.03131","n_code_links":0,"syntology":null},{"paper":"/paper/mvmoe-multi-task-vehicle-routing-solver-with","slug":"mvmoe-multi-task-vehicle-routing-solver-with","title":"MVMoE: Multi-Task Vehicle Routing Solver with Mixture-of-Experts","date":"2024-05-02","arxiv_id":"2405.01029","n_code_links":2,"syntology":{"ran":4,"of":5,"n_ran_checked":2,"n_instrument":2,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ai4co/awesome-fm4co","royalskye/routing-mvmoe"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"powering-in-database-dynamic-model-slicing","title":"Powering In-Database Dynamic Model Slicing for Structured Data Analytics","date":"2024-05-01","arxiv_id":"2405.00568","n_code_links":0,"syntology":null},{"paper":null,"slug":"lancet-accelerating-mixture-of-experts","title":"Lancet: Accelerating Mixture-of-Experts Training via Whole Graph Computation-Communication Overlapping","date":"2024-04-30","arxiv_id":"2404.19429","n_code_links":0,"syntology":null},{"paper":null,"slug":"integration-of-mixture-of-experts-and","title":"Integration of Mixture of Experts and Multimodal Generative AI in Internet of Vehicles: A Survey","date":"2024-04-25","arxiv_id":"2404.16356","n_code_links":0,"syntology":null},{"paper":null,"slug":"prediction-is-all-moe-needs-expert-load","title":"Prediction Is All MoE Needs: Expert Load Distribution Goes from Fluctuating to Stabilizing","date":"2024-04-25","arxiv_id":"2404.16914","n_code_links":0,"syntology":null},{"paper":null,"slug":"u2-moe-scaling-4-7x-parameters-with-minimal","title":"U2++ MoE: Scaling 4.7x parameters with minimal impact on RTF","date":"2024-04-25","arxiv_id":"2404.16407","n_code_links":0,"syntology":null},{"paper":"/paper/xft-unlocking-the-power-of-code-instruction","slug":"xft-unlocking-the-power-of-code-instruction","title":"XFT: Unlocking the Power of Code Instruction Tuning by Simply Merging Upcycled Mixture-of-Experts","date":"2024-04-23","arxiv_id":"2404.15247","n_code_links":1,"syntology":{"ran":11,"of":14,"n_ran_checked":9,"n_instrument":2,"unverified":3,"pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["ise-uiuc/xft"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/mixlora-enhancing-large-language-models-fine","slug":"mixlora-enhancing-large-language-models-fine","title":"MixLoRA: Enhancing Large Language Models Fine-Tuning with LoRA-based Mixture of Experts","date":"2024-04-22","arxiv_id":"2404.15159","n_code_links":2,"syntology":{"ran":6,"of":11,"n_ran_checked":6,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["TUDB-Labs/MixLoRA","mikecovlee/mLoRA"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/phi-3-technical-report-a-highly-capable","slug":"phi-3-technical-report-a-highly-capable","title":"Phi-3 Technical Report: A Highly Capable Language Model Locally on Your Phone","date":"2024-04-22","arxiv_id":"2404.14219","n_code_links":0,"syntology":null},{"paper":"/paper/moe-tinymed-mixture-of-experts-for-tiny","slug":"moe-tinymed-mixture-of-experts-for-tiny","title":"Med-MoE: Mixture of Domain-Specific Experts for Lightweight Medical Vision-Language Models","date":"2024-04-16","arxiv_id":"2404.10237","n_code_links":2,"syntology":{"ran":7,"of":12,"n_ran_checked":4,"n_instrument":3,"unverified":5,"pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","official":{"repos":["jiangsongtao/med-moe","jiangsongtao/tinymed"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"ctrl-adapter-an-efficient-and-versatile","title":"Ctrl-Adapter: An Efficient and Versatile Framework for Adapting Diverse Controls to Any Diffusion Model","date":"2024-04-15","arxiv_id":"2404.09967","n_code_links":0,"syntology":null},{"paper":"/paper/moe-ffd-mixture-of-experts-for-generalized","slug":"moe-ffd-mixture-of-experts-for-generalized","title":"MoE-FFD: Mixture of Experts for Generalized and Parameter-Efficient Face Forgery Detection","date":"2024-04-12","arxiv_id":"2404.08452","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":5,"n_instrument":1,"unverified":0,"pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lovesiamesecat/moe-ffd"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"dense-training-sparse-inference-rethinking","title":"Dense Training, Sparse Inference: Rethinking Training of Mixture-of-Experts Language Models","date":"2024-04-08","arxiv_id":"2404.05567","n_code_links":0,"syntology":null},{"paper":null,"slug":"physics-of-language-models-part-3-3-knowledge","title":"Physics of Language Models: Part 3.3, Knowledge Capacity Scaling Laws","date":"2024-04-08","arxiv_id":"2404.05405","n_code_links":0,"syntology":null},{"paper":null,"slug":"seer-moe-sparse-expert-efficiency-through","title":"SEER-MoE: Sparse Expert Efficiency through Regularization for Mixture-of-Experts","date":"2024-04-07","arxiv_id":"2404.05089","n_code_links":0,"syntology":null},{"paper":null,"slug":"shortcut-connected-expert-parallelism-for","title":"Shortcut-connected Expert Parallelism for Accelerating Mixture-of-Experts","date":"2024-04-07","arxiv_id":"2404.05019","n_code_links":0,"syntology":null},{"paper":null,"slug":"toward-inference-optimal-mixture-of-expert","title":"Toward Inference-optimal Mixture-of-Expert Large Language Models","date":"2024-04-03","arxiv_id":"2404.02852","n_code_links":0,"syntology":null},{"paper":"/paper/two-heads-are-better-than-one-nested-poe-for","slug":"two-heads-are-better-than-one-nested-poe-for","title":"Two Heads are Better than One: Nested PoE for Robust Defense Against Multi-Backdoors","date":"2024-04-02","arxiv_id":"2404.02356","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["victoriagraf/nested_poe"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/prompt-prompted-mixture-of-experts-for","slug":"prompt-prompted-mixture-of-experts-for","title":"Prompt-prompted Adaptive Structured Pruning for Efficient LLM Generation","date":"2024-04-01","arxiv_id":"2404.01365","n_code_links":1,"syntology":{"ran":9,"of":10,"n_ran_checked":7,"n_instrument":2,"unverified":1,"pointer_only":10,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["hdong920/griffin"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/jamba-a-hybrid-transformer-mamba-language","slug":"jamba-a-hybrid-transformer-mamba-language","title":"Jamba: A Hybrid Transformer-Mamba Language Model","date":"2024-03-28","arxiv_id":"2403.19887","n_code_links":3,"syntology":null},{"paper":"/paper/mini-gemini-mining-the-potential-of-multi","slug":"mini-gemini-mining-the-potential-of-multi","title":"Mini-Gemini: Mining the Potential of Multi-modality Vision Language Models","date":"2024-03-27","arxiv_id":"2403.18814","n_code_links":2,"syntology":{"ran":8,"of":8,"n_ran_checked":5,"n_instrument":3,"unverified":0,"pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 1 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["dvlab-research/minigemini"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"generalization-error-analysis-for-sparse","title":"Generalization Error Analysis for Sparse Mixture-of-Experts: A Preliminary Study","date":"2024-03-26","arxiv_id":"2403.17404","n_code_links":0,"syntology":null},{"paper":"/paper/multi-task-dense-prediction-via-mixture-of","slug":"multi-task-dense-prediction-via-mixture-of","title":"Multi-Task Dense Prediction via Mixture of Low-Rank Experts","date":"2024-03-26","arxiv_id":"2403.17749","n_code_links":1,"syntology":{"ran":9,"of":11,"n_ran_checked":7,"n_instrument":2,"unverified":2,"pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["yuqiyang213/mlore"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/tiny-models-are-the-computational-saver-for","slug":"tiny-models-are-the-computational-saver-for","title":"Tiny Models are the Computational Saver for Large Models","date":"2024-03-26","arxiv_id":"2403.17726","n_code_links":1,"syntology":null},{"paper":"/paper/boosting-continual-learning-of-vision","slug":"boosting-continual-learning-of-vision","title":"Boosting Continual Learning of Vision-Language Models via Mixture-of-Experts Adapters","date":"2024-03-18","arxiv_id":"2403.11549","n_code_links":2,"syntology":{"ran":2,"of":4,"n_ran_checked":1,"n_instrument":1,"unverified":2,"pointer_only":4,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["jiazuoyu/moe-adapters4cl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/mope-parameter-efficient-and-scalable","slug":"mope-parameter-efficient-and-scalable","title":"MoPE: Mixture of Prompt Experts for Parameter-Efficient and Scalable Multimodal Fusion","date":"2024-03-14","arxiv_id":"2403.10568","n_code_links":1,"syntology":null},{"paper":"/paper/branch-train-mix-mixing-expert-llms-into-a","slug":"branch-train-mix-mixing-expert-llms-into-a","title":"Branch-Train-MiX: Mixing Expert LLMs into a Mixture-of-Experts LLM","date":"2024-03-12","arxiv_id":"2403.07816","n_code_links":1,"syntology":null},{"paper":"/paper/equipping-computational-pathology-systems","slug":"equipping-computational-pathology-systems","title":"Equipping Computational Pathology Systems with Artifact Processing Pipelines: A Showcase for Computation and Performance Trade-offs","date":"2024-03-12","arxiv_id":"2403.07743","n_code_links":1,"syntology":null},{"paper":"/paper/harder-tasks-need-more-experts-dynamic","slug":"harder-tasks-need-more-experts-dynamic","title":"Harder Tasks Need More Experts: Dynamic Routing in MoE Models","date":"2024-03-12","arxiv_id":"2403.07652","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":0,"n_instrument":4,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zhenweian/dynamic_moe"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/designing-effective-sparse-expert-models","slug":"designing-effective-sparse-expert-models","title":"ST-MoE: Designing Stable and Transferable Sparse Expert Models","date":"2022-02-17","arxiv_id":"2202.08906","n_code_links":3,"syntology":{"ran":5,"of":5,"n_ran_checked":3,"n_instrument":2,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tensorflow/mesh"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}}],"record_sha256":"87efb09214d5eccf8cf86fe1e18b45d814e7d540a4c7761054317fb035c50823","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}