{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/mixture-of-experts/papers/11","list_of":"/task/mixture-of-experts","task":"Mixture-of-Experts","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":11,"pages_in_order":14,"rows_per_page":100,"rows":[1001,1100],"of":1312,"counts":{"archive_papers_tagged":1312,"with_a_code_link":516,"where_syntology_ran_a_sample":216,"not_listed_spam_title":0,"listed":1312,"listed_where_code_ran":216,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":184,"every_run_a_failure_of_syntologys_instrument":32,"listed_with_a_run_with_no_instrument_failure":184,"listed_every_run_a_failure_of_syntologys_instrument":32,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/mixture-of-experts","prev":"/task/mixture-of-experts/papers/10","next":"/task/mixture-of-experts/papers/12","papers":[{"url":null,"slug":"half-space-feature-learning-in-neural","title":"Half-Space Feature Learning in Neural Networks","date":"2024-04-05","arxiv_id":"2404.04312","repositories_listed":0,"syntology":null},{"url":null,"slug":"psychometry-an-omnifit-model-for-image","title":"Psychometry: An Omnifit Model for Image Reconstruction from Human Brain Activity","date":"2024-03-29","arxiv_id":"2403.20022","repositories_listed":0,"syntology":null},{"url":null,"slug":"revolutionizing-disease-diagnosis-with","title":"Revolutionizing Disease Diagnosis with simultaneous functional PET/MR and Deeply Integrated Brain Metabolic, Hemodynamic, and Perfusion Networks","date":"2024-03-29","arxiv_id":"2403.20058","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalization-error-analysis-for-sparse","title":"Generalization Error Analysis for Sparse Mixture-of-Experts: A Preliminary Study","date":"2024-03-26","arxiv_id":"2403.17404","repositories_listed":0,"syntology":null},{"url":null,"slug":"germ-a-generalist-robotic-model-with-mixture","title":"GeRM: A Generalist Robotic Model with Mixture-of-experts for Quadruped Robot","date":"2024-03-20","arxiv_id":"2403.13358","repositories_listed":0,"syntology":null},{"url":"/paper/mm1-methods-analysis-insights-from-multimodal","slug":"mm1-methods-analysis-insights-from-multimodal","title":"MM1: Methods, Analysis & Insights from Multimodal LLM Pre-training","date":"2024-03-14","arxiv_id":"2403.09611","repositories_listed":0,"syntology":null},{"url":null,"slug":"conditional-computation-in-neural-networks-1","title":"Conditional computation in neural networks: principles and research trends","date":"2024-03-12","arxiv_id":"2403.07965","repositories_listed":0,"syntology":null},{"url":null,"slug":"acquiring-diverse-skills-using-curriculum","title":"Acquiring Diverse Skills using Curriculum Reinforcement Learning with Mixture of Experts","date":"2024-03-11","arxiv_id":"2403.06966","repositories_listed":0,"syntology":null},{"url":null,"slug":"mmoe-robust-spoiler-detection-with-multi","title":"MMoE: Robust Spoiler Detection with Multi-modal Information and Domain-aware Mixture-of-Experts","date":"2024-03-08","arxiv_id":"2403.05265","repositories_listed":0,"syntology":null},{"url":null,"slug":"constitutionalexperts-training-a-mixture-of","title":"ConstitutionalExperts: Training a Mixture of Principle-based Prompts","date":"2024-03-07","arxiv_id":"2403.04894","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-does-architecture-influence-the-base","title":"How does Architecture Influence the Base Capabilities of Pre-trained Language Models? A Case Study Based on FFN-Wider and MoE Transformers","date":"2024-03-04","arxiv_id":"2403.02436","repositories_listed":0,"syntology":null},{"url":null,"slug":"hypertext-entity-extraction-in-webpage","title":"Hypertext Entity Extraction in Webpage","date":"2024-03-04","arxiv_id":"2403.01698","repositories_listed":0,"syntology":null},{"url":null,"slug":"vanilla-transformers-are-transfer-capability","title":"Vanilla Transformers are Transfer Capability Teachers","date":"2024-03-04","arxiv_id":"2403.01994","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-the-immunity-of-mixture-of-experts","title":"Enhancing the \"Immunity\" of Mixture-of-Experts Networks for Adversarial Defense","date":"2024-02-29","arxiv_id":"2402.18787","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-effective-mixture-of-experts-approach-for","title":"An Effective Mixture-Of-Experts Approach For Code-Switching Speech Recognition Leveraging Encoder Disentanglement","date":"2024-02-27","arxiv_id":"2402.17189","repositories_listed":0,"syntology":null},{"url":null,"slug":"co-supervised-learning-improving-weak-to","title":"Co-Supervised Learning: Improving Weak-to-Strong Generalization with Hierarchical Mixture of Experts","date":"2024-02-23","arxiv_id":"2402.15505","repositories_listed":0,"syntology":null},{"url":null,"slug":"pemt-multi-task-correlation-guided-mixture-of","title":"PEMT: Multi-Task Correlation Guided Mixture-of-Experts Enables Parameter-Efficient Transfer Learning","date":"2024-02-23","arxiv_id":"2402.15082","repositories_listed":0,"syntology":null},{"url":null,"slug":"denoising-oct-images-using-steered-mixture-of","title":"Denoising OCT Images Using Steered Mixture of Experts with Multi-Model Inference","date":"2024-02-20","arxiv_id":"2402.12735","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-an-empirical-understanding-of-moe","title":"Towards an empirical understanding of MoE design choices","date":"2024-02-20","arxiv_id":"2402.13089","repositories_listed":0,"syntology":null},{"url":null,"slug":"moral-moe-augmented-lora-for-llms-lifelong","title":"MoRAL: MoE Augmented LoRA for LLMs' Lifelong Learning","date":"2024-02-17","arxiv_id":"2402.11260","repositories_listed":0,"syntology":null},{"url":null,"slug":"turn-waste-into-worth-rectifying-top-k-router","title":"Turn Waste into Worth: Rectifying Top-$k$ Router of MoE","date":"2024-02-17","arxiv_id":"2402.12399","repositories_listed":0,"syntology":null},{"url":null,"slug":"amend-a-mixture-of-experts-framework-for-long","title":"AMEND: A Mixture of Experts Framework for Long-tailed Trajectory Prediction","date":"2024-02-13","arxiv_id":"2402.08698","repositories_listed":0,"syntology":null},{"url":null,"slug":"p-mamba-marrying-perona-malik-diffusion-with","title":"P-Mamba: Marrying Perona Malik Diffusion with Mamba for Efficient Pediatric Echocardiographic Left Ventricular Segmentation","date":"2024-02-13","arxiv_id":"2402.08506","repositories_listed":0,"syntology":null},{"url":null,"slug":"differentially-private-training-of-mixture-of","title":"Differentially Private Training of Mixture of Experts Models","date":"2024-02-11","arxiv_id":"2402.07334","repositories_listed":0,"syntology":null},{"url":null,"slug":"buffer-overflow-in-mixture-of-experts","title":"Buffer Overflow in Mixture of Experts","date":"2024-02-08","arxiv_id":"2402.05526","repositories_listed":0,"syntology":null},{"url":null,"slug":"task-customized-masked-autoencoder-via","title":"Task-customized Masked AutoEncoder via Mixture of Cluster-conditional Experts","date":"2024-02-08","arxiv_id":"2402.05382","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-parameter-estimation-in-deviated-gaussian","title":"On Parameter Estimation in Deviated Gaussian Mixture of Experts","date":"2024-02-07","arxiv_id":"2402.05220","repositories_listed":0,"syntology":null},{"url":null,"slug":"breaking-the-curse-of-dimensionality-with-2","title":"Approximation Rates and VC-Dimension Bounds for (P)ReLU MLP Mixture of Experts","date":"2024-02-05","arxiv_id":"2402.03460","repositories_listed":0,"syntology":null},{"url":"/paper/fusemoe-mixture-of-experts-transformers-for","slug":"fusemoe-mixture-of-experts-transformers-for","title":"FuseMoE: Mixture-of-Experts Transformers for Fleximodal Fusion","date":"2024-02-05","arxiv_id":"2402.03226","repositories_listed":0,"syntology":{"n":4,"n_ran":3,"n_constructed":1,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/fusemoe-mixture-of-experts-transformers-for#ran","syntology_url":"https://syntology.ai/paper/2402.03226","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.03226"}},"official":null}},{"url":null,"slug":"on-least-squares-estimation-in-softmax-gating","title":"On Least Square Estimation in Softmax Gating Mixture of Experts","date":"2024-02-05","arxiv_id":"2402.02952","repositories_listed":0,"syntology":null},{"url":null,"slug":"mode-a-mixture-of-experts-model-with-mutual","title":"MoDE: A Mixture-of-Experts Model with Mutual Distillation among the Experts","date":"2024-01-31","arxiv_id":"2402.00893","repositories_listed":0,"syntology":null},{"url":null,"slug":"explainable-data-driven-modeling-via-mixture","title":"Explainable data-driven modeling via mixture of experts: towards effective blending of grey and black-box models","date":"2024-01-30","arxiv_id":"2401.17118","repositories_listed":0,"syntology":null},{"url":null,"slug":"llava-mole-sparse-mixture-of-lora-experts-for","title":"LLaVA-MoLE: Sparse Mixture of LoRA Experts for Mitigating Data Conflicts in Instruction Finetuning MLLMs","date":"2024-01-29","arxiv_id":"2401.16160","repositories_listed":0,"syntology":null},{"url":null,"slug":"routers-in-vision-mixture-of-experts-an","title":"Routers in Vision Mixture of Experts: An Empirical Study","date":"2024-01-29","arxiv_id":"2401.15969","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-temperature-sample-efficient-for-softmax","title":"Is Temperature Sample Efficient for Softmax Gaussian Mixture of Experts?","date":"2024-01-25","arxiv_id":"2401.13875","repositories_listed":0,"syntology":null},{"url":null,"slug":"m-3-tn-multi-gate-mixture-of-experts-based","title":"M$^3$TN: Multi-gate Mixture-of-Experts based Multi-valued Treatment Network for Uplift Modeling","date":"2024-01-24","arxiv_id":"2401.14426","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-better-metric-for-text-to-video","title":"Towards A Better Metric for Text-to-Video Generation","date":"2024-01-15","arxiv_id":"2401.07781","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompt-based-mental-health-screening-from","title":"Prompt-based mental health screening from social media text","date":"2024-01-11","arxiv_id":"2401.05912","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-calibration-for-improved-weather","title":"Robust Calibration For Improved Weather Prediction Under Distributional Shift","date":"2024-01-08","arxiv_id":"2401.04144","repositories_listed":0,"syntology":null},{"url":null,"slug":"incorporating-visual-experts-to-resolve-the","title":"Incorporating Visual Experts to Resolve the Information Loss in Multimodal Large Language Models","date":"2024-01-06","arxiv_id":"2401.03105","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-deweather-mixture-of-experts-with","title":"Efficient Deweather Mixture-of-Experts with Uncertainty-aware Feature-wise Linear Modulation","date":"2023-12-27","arxiv_id":"2312.16610","repositories_listed":0,"syntology":null},{"url":null,"slug":"agent4ranking-semantic-robust-ranking-via","title":"Agent4Ranking: Semantic Robust Ranking via Personalized Query Rewriting Using Multi-agent LLM","date":"2023-12-24","arxiv_id":"2312.15450","repositories_listed":0,"syntology":null},{"url":null,"slug":"generator-assisted-mixture-of-experts-for","title":"Generator Assisted Mixture of Experts For Feature Acquisition in Batch","date":"2023-12-19","arxiv_id":"2312.12574","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-of-cluster-conditional-lora-experts","title":"Mixture of Cluster-conditional LoRA Experts for Vision-language Instruction Tuning","date":"2023-12-19","arxiv_id":"2312.12379","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-google-gemini-to-openai-q-q-star-a","title":"From Google Gemini to OpenAI Q* (Q-Star): A Survey of Reshaping the Generative Artificial Intelligence (AI) Research Landscape","date":"2023-12-18","arxiv_id":"2312.10868","repositories_listed":0,"syntology":null},{"url":null,"slug":"training-of-neural-networks-with-uncertain","title":"Training of Neural Networks with Uncertain Data: A Mixture of Experts Approach","date":"2023-12-13","arxiv_id":"2312.08083","repositories_listed":0,"syntology":null},{"url":null,"slug":"moe-amc-enhancing-automatic-modulation","title":"MoE-AMC: Enhancing Automatic Modulation Classification Performance Using Mixture-of-Experts","date":"2023-12-04","arxiv_id":"2312.02298","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-driven-all-in-one-adverse-weather","title":"Language-driven All-in-one Adverse Weather Removal","date":"2023-12-03","arxiv_id":"2312.01381","repositories_listed":0,"syntology":null},{"url":null,"slug":"moec-mixture-of-experts-implicit-neural","title":"MoEC: Mixture of Experts Implicit Neural Compression","date":"2023-12-03","arxiv_id":"2312.01361","repositories_listed":0,"syntology":null},{"url":"/paper/omni-smola-boosting-generalist-multimodal","slug":"omni-smola-boosting-generalist-multimodal","title":"Omni-SMoLA: Boosting Generalist Multimodal Models with Soft Mixture of Low-rank Experts","date":"2023-12-01","arxiv_id":"2312.00968","repositories_listed":0,"syntology":null},{"url":null,"slug":"homoe-a-memory-based-and-composition-aware","title":"HOMOE: A Memory-Based and Composition-Aware Framework for Zero-Shot Learning with Hopfield Network and Soft Mixture of Experts","date":"2023-11-23","arxiv_id":"2311.14747","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-model-agnostic-approach-for","title":"Efficient Model Agnostic Approach for Implicit Neural Representation Based Arbitrary-Scale Image Super-Resolution","date":"2023-11-20","arxiv_id":"2311.12077","repositories_listed":0,"syntology":null},{"url":null,"slug":"memory-augmented-language-models-through","title":"Memory Augmented Language Models through Mixture of Word Experts","date":"2023-11-15","arxiv_id":"2311.10768","repositories_listed":0,"syntology":null},{"url":null,"slug":"intentional-biases-in-llm-responses","title":"Intentional Biases in LLM Responses","date":"2023-11-11","arxiv_id":"2311.07611","repositories_listed":0,"syntology":null},{"url":null,"slug":"came-competitively-learning-a-mixture-of","title":"CAME: Competitively Learning a Mixture-of-Experts Model for First-stage Retrieval","date":"2023-11-06","arxiv_id":"2311.02834","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-of-experts-for-open-set-domain","title":"Mixture-of-Experts for Open Set Domain Adaptation: A Dual-Space Detection Approach","date":"2023-11-01","arxiv_id":"2311.00285","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-general-theory-for-softmax-gating","title":"A General Theory for Softmax Gating Multinomial Logistic Mixture of Experts","date":"2023-10-22","arxiv_id":"2310.14188","repositories_listed":0,"syntology":null},{"url":null,"slug":"direct-neural-machine-translation-with-task","title":"Direct Neural Machine Translation with Task-level Mixture of Experts models","date":"2023-10-18","arxiv_id":"2310.12236","repositories_listed":0,"syntology":null},{"url":null,"slug":"diversifying-the-mixture-of-experts","title":"Diversifying the Mixture-of-Experts Representation for Language Models with Orthogonal Optimizer","date":"2023-10-15","arxiv_id":"2310.09762","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-gating-in-mixture-of-experts-based","title":"Adaptive Gating in Mixture-of-Experts based Language Models","date":"2023-10-11","arxiv_id":"2310.07188","repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-non-uniform-uncertainty-in-reaction","title":"Beyond the Typical: Modeling Rare Plausible Patterns in Chemical Reactions by Leveraging Sequential Mixture-of-Experts","date":"2023-10-07","arxiv_id":"2310.04674","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-mixture-of","title":"Reinforcement Learning-based Mixture of Vision Transformers for Video Violence Recognition","date":"2023-10-04","arxiv_id":"2310.03108","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-of-quantized-experts-moqe","title":"Mixture of Quantized Experts (MoQE): Complementary Effect of Low-bit Quantization and Robustness","date":"2023-10-03","arxiv_id":"2310.02410","repositories_listed":0,"syntology":null},{"url":null,"slug":"statistical-perspective-of-top-k-sparse","title":"Statistical Perspective of Top-K Sparse Softmax Gating Mixture of Experts","date":"2023-09-25","arxiv_id":"2309.13850","repositories_listed":0,"syntology":null},{"url":null,"slug":"mobile-v-moes-scaling-down-vision","title":"Mobile V-MoEs: Scaling Down Vision Transformers via Sparse Mixture-of-Experts","date":"2023-09-08","arxiv_id":"2309.04354","repositories_listed":0,"syntology":null},{"url":null,"slug":"task-based-moe-for-multitask-multilingual","title":"Task-Based MoE for Multitask Multilingual Machine Translation","date":"2023-08-30","arxiv_id":"2308.15772","repositories_listed":0,"syntology":null},{"url":null,"slug":"serving-moe-models-on-resource-constrained","title":"SwapMoE: Serving Off-the-shelf MoE-based Large Language Models with Tunable Memory Budget","date":"2023-08-29","arxiv_id":"2308.15030","repositories_listed":0,"syntology":null},{"url":null,"slug":"eve-efficient-vision-language-pre-training","title":"EVE: Efficient Vision-Language Pre-training with Masked Prediction and Modality-Aware MoE","date":"2023-08-23","arxiv_id":"2308.11971","repositories_listed":0,"syntology":null},{"url":null,"slug":"finequant-unlocking-efficiency-with-fine","title":"FineQuant: Unlocking Efficiency with Fine-Grained Weight-Only Quantization for LLMs","date":"2023-08-16","arxiv_id":"2308.09723","repositories_listed":0,"syntology":null},{"url":null,"slug":"experts-weights-averaging-a-new-general","title":"Experts Weights Averaging: A New General Training Scheme for Vision Transformers","date":"2023-08-11","arxiv_id":"2308.06093","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-temporal-multi-gate-mixture-of","title":"A Novel Temporal Multi-Gate Mixture-of-Experts Approach for Vehicle Trajectory and Driving Intention Prediction","date":"2023-08-01","arxiv_id":"2308.00533","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-encoded-multi-modal-fusion-for","title":"Uncertainty-Encoded Multi-Modal Fusion for Robust Object Detection in Autonomous Driving","date":"2023-07-30","arxiv_id":"2307.16121","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-efficient-general-purpose-modular-vision","title":"An Efficient General-Purpose Modular Vision Model via Multi-Task Heterogeneous Training","date":"2023-06-29","arxiv_id":"2306.17165","repositories_listed":0,"syntology":null},{"url":null,"slug":"skillnet-x-a-multilingual-multitask-model","title":"SkillNet-X: A Multilingual Multitask Model with Sparsely Activated Skills","date":"2023-06-28","arxiv_id":"2306.16176","repositories_listed":0,"syntology":null},{"url":null,"slug":"jiuzhang-2-0-a-unified-chinese-pre-trained","title":"JiuZhang 2.0: A Unified Chinese Pre-trained Language Model for Multi-task Mathematical Problem Solving","date":"2023-06-19","arxiv_id":"2306.11027","repositories_listed":0,"syntology":null},{"url":null,"slug":"fed-zero-efficient-zero-shot-personalization","title":"Learning to Specialize: Joint Gating-Expert Training for Adaptive MoEs in Decentralized Settings","date":"2023-06-14","arxiv_id":"2306.08586","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-weighted-mixture-of-experts-with","title":"Attention Weighted Mixture of Experts with Contrastive Learning for Personalized Ranking in E-commerce","date":"2023-06-08","arxiv_id":"2306.05011","repositories_listed":0,"syntology":null},{"url":null,"slug":"divide-conquer-and-combine-mixture-of","title":"Divide, Conquer, and Combine: Mixture of Semantic-Independent Experts for Zero-Shot Dialogue State Tracking","date":"2023-06-01","arxiv_id":"2306.00434","repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-task-relationships-in-multi-variate","title":"Modeling Task Relationships in Multi-variate Soft Sensor with Balanced Mixture-of-Experts","date":"2023-05-25","arxiv_id":"2305.16360","repositories_listed":0,"syntology":null},{"url":null,"slug":"flan-moe-scaling-instruction-finetuned","title":"Mixture-of-Experts Meets Instruction Tuning:A Winning Combination for Large Language Models","date":"2023-05-24","arxiv_id":"2305.14705","repositories_listed":0,"syntology":null},{"url":null,"slug":"pre-training-multi-task-contrastive-learning","title":"Pre-training Multi-task Contrastive Learning Models for Scientific Literature Understanding","date":"2023-05-23","arxiv_id":"2305.14232","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-unified-view-of-sparse-feed-forward","title":"Towards A Unified View of Sparse Feed-Forward Network in Pretraining Large Language Model","date":"2023-05-23","arxiv_id":"2305.13999","repositories_listed":0,"syntology":null},{"url":null,"slug":"to-repeat-or-not-to-repeat-insights-from","title":"To Repeat or Not To Repeat: Insights from Scaling LLM under Token-Crisis","date":"2023-05-22","arxiv_id":"2305.13230","repositories_listed":0,"syntology":null},{"url":null,"slug":"lifelong-language-pretraining-with","title":"Lifelong Language Pretraining with Distribution-Specialized Experts","date":"2023-05-20","arxiv_id":"2305.12281","repositories_listed":0,"syntology":null},{"url":null,"slug":"locking-and-quacking-stacking-bayesian-model","title":"Locking and Quacking: Stacking Bayesian model predictions by log-pooling and superposition","date":"2023-05-12","arxiv_id":"2305.07334","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-convergence-rates-for-parameter","title":"Towards Convergence Rates for Parameter Estimation in Gaussian-gated Mixture of Experts","date":"2023-05-12","arxiv_id":"2305.07572","repositories_listed":0,"syntology":null},{"url":"/paper/alternating-gradient-descent-and-mixture-of","slug":"alternating-gradient-descent-and-mixture-of","title":"Alternating Gradient Descent and Mixture-of-Experts for Integrated Multimodal Perception","date":"2023-05-10","arxiv_id":"2305.06324","repositories_listed":0,"syntology":null},{"url":null,"slug":"demystifying-softmax-gating-in-gaussian","title":"Demystifying Softmax Gating Function in Gaussian Mixture of Experts","date":"2023-05-05","arxiv_id":"2305.03288","repositories_listed":0,"syntology":null},{"url":null,"slug":"steered-mixture-of-experts-autoencoder-design","title":"Steered Mixture-of-Experts Autoencoder Design for Real-Time Image Modelling and Denoising","date":"2023-05-05","arxiv_id":"2305.03485","repositories_listed":0,"syntology":null},{"url":null,"slug":"pipeline-moe-a-flexible-moe-implementation","title":"Pipeline MoE: A Flexible MoE Implementation with Pipeline Parallelism","date":"2023-04-22","arxiv_id":"2304.11414","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-single-gated-mixtures-of-experts","title":"Revisiting Single-gated Mixtures of Experts","date":"2023-04-11","arxiv_id":"2304.05497","repositories_listed":0,"syntology":null},{"url":null,"slug":"flexmoe-scaling-large-scale-sparse-pre","title":"FlexMoE: Scaling Large-scale Sparse Pre-trained Model Training via Dynamic Device Placement","date":"2023-04-08","arxiv_id":"2304.03946","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixed-regression-via-approximate-message","title":"Mixed Regression via Approximate Message Passing","date":"2023-04-05","arxiv_id":"2304.02229","repositories_listed":0,"syntology":null},{"url":null,"slug":"steered-mixture-of-experts-regression-for","title":"Steered Mixture of Experts Regression for Image Denoising with Multi-Model-Inference","date":"2023-03-30","arxiv_id":"2303.17409","repositories_listed":0,"syntology":null},{"url":null,"slug":"mowe-mixture-of-weather-experts-for-multiple","title":"WM-MoE: Weather-aware Multi-scale Mixture-of-Experts for Blind Adverse Weather Removal","date":"2023-03-24","arxiv_id":"2303.13739","repositories_listed":0,"syntology":null},{"url":null,"slug":"disguise-without-disruption-utility","title":"Disguise without Disruption: Utility-Preserving Face De-Identification","date":"2023-03-23","arxiv_id":"2303.13269","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-small-scale-switch-transformer-and-nlp","title":"Improving Transformer Performance for French Clinical Notes Classification Using Mixture of Experts on a Limited Dataset","date":"2023-03-22","arxiv_id":"2303.12892","repositories_listed":0,"syntology":null},{"url":null,"slug":"hdformer-a-higher-dimensional-transformer-for","title":"HDformer: A Higher Dimensional Transformer for Diabetes Detection Utilizing Long Range Vascular Signals","date":"2023-03-17","arxiv_id":"2303.11340","repositories_listed":0,"syntology":null},{"url":null,"slug":"mcr-dl-mix-and-match-communication-runtime","title":"MCR-DL: Mix-and-Match Communication Runtime for Deep Learning","date":"2023-03-15","arxiv_id":"2303.08374","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-vision-language-models-with-sparse","title":"Scaling Vision-Language Models with Sparse Mixture of Experts","date":"2023-03-13","arxiv_id":"2303.07226","repositories_listed":0,"syntology":null}],"record_sha256":"d695b827197dc65dd69077a5dbf0f69c5440070732e9ad9d79ba7a134ef8c230","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}