{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/mixture-of-experts/papers/10","list_of":"/task/mixture-of-experts","task":"Mixture-of-Experts","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":10,"pages_in_order":14,"rows_per_page":100,"rows":[901,1000],"of":1312,"counts":{"archive_papers_tagged":1312,"with_a_code_link":516,"where_syntology_ran_a_sample":216,"not_listed_spam_title":0,"listed":1312,"listed_where_code_ran":216,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":184,"every_run_a_failure_of_syntologys_instrument":32,"listed_with_a_run_with_no_instrument_failure":184,"listed_every_run_a_failure_of_syntologys_instrument":32,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/mixture-of-experts","prev":"/task/mixture-of-experts/papers/9","next":"/task/mixture-of-experts/papers/11","papers":[{"url":null,"slug":"interpretable-mixture-of-experts-for-time","title":"Interpretable mixture of experts for time series prediction under recurrent and non-recurrent conditions","date":"2024-09-05","arxiv_id":"2409.03282","repositories_listed":0,"syntology":null},{"url":null,"slug":"configurable-foundation-models-building-llms","title":"Configurable Foundation Models: Building LLMs from a Modular Perspective","date":"2024-09-04","arxiv_id":"2409.02877","repositories_listed":0,"syntology":null},{"url":null,"slug":"pluralistic-salient-object-detection","title":"Pluralistic Salient Object Detection","date":"2024-09-04","arxiv_id":"2409.02368","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-code-switching-speech-recognition-1","title":"Enhancing Code-Switching Speech Recognition with LID-Based Collaborative Mixture of Experts Model","date":"2024-09-03","arxiv_id":"2409.02050","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-parameter-count-implicit-bias-in-soft","title":"Beyond Parameter Count: Implicit Bias in Soft Mixture of Experts","date":"2024-09-02","arxiv_id":"2409.00879","repositories_listed":0,"syntology":null},{"url":null,"slug":"duplex-a-device-for-large-language-models","title":"Duplex: A Device for Large Language Models with Mixture of Experts, Grouped Query Attention, and Continuous Batching","date":"2024-09-02","arxiv_id":"2409.01141","repositories_listed":0,"syntology":null},{"url":null,"slug":"auxiliary-loss-free-load-balancing-strategy","title":"Auxiliary-Loss-Free Load Balancing Strategy for Mixture-of-Experts","date":"2024-08-28","arxiv_id":"2408.15664","repositories_listed":0,"syntology":null},{"url":null,"slug":"nexus-specialization-meets-adaptability-for","title":"Nexus: Specialization meets Adaptability for Efficiently Training Mixture of Experts","date":"2024-08-28","arxiv_id":"2408.15901","repositories_listed":0,"syntology":null},{"url":null,"slug":"parameter-efficient-quantized-mixture-of","title":"Parameter-Efficient Quantized Mixture-of-Experts Meets Vision-Language Instruction Tuning for Semiconductor Electron Micrograph Analysis","date":"2024-08-27","arxiv_id":"2408.15305","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-enterprise-spatio-temporal","title":"Advancing Enterprise Spatio-Temporal Forecasting Applications: Data Mining Meets Instruction Tuning of Language Models For Multi-modal Time Series Analysis in Low-Resource Settings","date":"2024-08-24","arxiv_id":"2408.13622","repositories_listed":0,"syntology":null},{"url":null,"slug":"la-softmoe-clip-for-unified-physical-digital","title":"La-SoftMoE CLIP for Unified Physical-Digital Face Attack Detection","date":"2024-08-23","arxiv_id":"2408.12793","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-treatment-multi-task-uplift-modeling","title":"Multi-Treatment Multi-Task Uplift Modeling for Enhancing User Growth","date":"2024-08-23","arxiv_id":"2408.12803","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-ultimate-guide-to-fine-tuning-llms-from","title":"The Ultimate Guide to Fine-Tuning LLMs from Basics to Breakthroughs: An Exhaustive Review of Technologies, Research, Best Practices, Applied Research Challenges and Opportunities","date":"2024-08-23","arxiv_id":"2408.13296","repositories_listed":0,"syntology":null},{"url":null,"slug":"sql-gen-bridging-the-dialect-gap-for-text-to","title":"SQL-GEN: Bridging the Dialect Gap for Text-to-SQL Via Synthetic Data And Model Merging","date":"2024-08-22","arxiv_id":"2408.12733","repositories_listed":0,"syntology":null},{"url":null,"slug":"fedmoe-personalized-federated-learning-via","title":"FedMoE: Personalized Federated Learning via Heterogeneous Mixture of Experts","date":"2024-08-21","arxiv_id":"2408.11304","repositories_listed":0,"syntology":null},{"url":null,"slug":"hmoe-heterogeneous-mixture-of-experts-for","title":"HMoE: Heterogeneous Mixture of Experts for Language Modeling","date":"2024-08-20","arxiv_id":"2408.10681","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-framework-for-iris-anti-spoofing","title":"A Unified Framework for Iris Anti-Spoofing: Introducing IrisGeneral Dataset and Masked-MoE Method","date":"2024-08-19","arxiv_id":"2408.09752","repositories_listed":0,"syntology":null},{"url":null,"slug":"bam-just-like-that-simple-and-efficient","title":"BAM! Just Like That: Simple and Efficient Parameter Upcycling for Mixture of Experts","date":"2024-08-15","arxiv_id":"2408.08274","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-model-moerging-recycling-and","title":"A Survey on Model MoErging: Recycling and Routing Among Specialized Experts for Collaborative Learning","date":"2024-08-13","arxiv_id":"2408.07057","repositories_listed":0,"syntology":null},{"url":null,"slug":"home-hierarchy-of-multi-gate-experts-for","title":"HoME: Hierarchy of Multi-Gate Experts for Multi-Task Learning at Kuaishou","date":"2024-08-10","arxiv_id":"2408.05430","repositories_listed":0,"syntology":null},{"url":null,"slug":"ladimo-layer-wise-distillation-inspired","title":"LaDiMo: Layer-wise Distillation Inspired MoEfier","date":"2024-08-08","arxiv_id":"2408.04278","repositories_listed":0,"syntology":null},{"url":null,"slug":"partial-experts-checkpoint-efficient-fault","title":"MoC-System: Efficient Fault Tolerance for Sparse Mixture-of-Experts Model Training","date":"2024-08-08","arxiv_id":"2408.04307","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-02306","title":"Mixture-of-Noises Enhanced Forgery-Aware Predictor for Multi-Face Manipulation Detection and Localization","date":"2024-08-05","arxiv_id":"2408.02306","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-01332","title":"HMDN: Hierarchical Multi-Distribution Network for Click-Through Rate Prediction","date":"2024-08-02","arxiv_id":"2408.01332","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-00365","title":"Multimodal Fusion and Coherence Modeling for Video Topic Segmentation","date":"2024-08-01","arxiv_id":"2408.00365","repositories_listed":0,"syntology":null},{"url":null,"slug":"2407-21571","title":"PMoE: Progressive Mixture of Experts with Asymmetric Transformer for Continual Learning","date":"2024-07-31","arxiv_id":"2407.21571","repositories_listed":0,"syntology":null},{"url":null,"slug":"moma-efficient-early-fusion-pre-training-with","title":"MoMa: Efficient Early-Fusion Pre-training with Mixture of Modality-Aware Experts","date":"2024-07-31","arxiv_id":"2407.21770","repositories_listed":0,"syntology":null},{"url":null,"slug":"distribution-learning-for-molecular","title":"Distribution Learning for Molecular Regression","date":"2024-07-30","arxiv_id":"2407.20475","repositories_listed":0,"syntology":null},{"url":null,"slug":"time-series-forecasting-with-high-stakes-a","title":"Time series forecasting with high stakes: A field study of the air cargo industry","date":"2024-07-29","arxiv_id":"2407.20192","repositories_listed":0,"syntology":null},{"url":null,"slug":"wolf-captioning-everything-with-a-world","title":"Wolf: Captioning Everything with a World Summarization Framework","date":"2024-07-26","arxiv_id":"2407.18908","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-lightweight-can-a-vision-transformer-be","title":"How Lightweight Can A Vision Transformer Be","date":"2024-07-25","arxiv_id":"2407.17783","repositories_listed":0,"syntology":null},{"url":null,"slug":"cheems-wonderful-matrices-more-efficient-and","title":"Wonderful Matrices: More Efficient and Effective Architecture for Language Modeling Tasks","date":"2024-07-24","arxiv_id":"2407.16958","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-domain-robust-lightweight-reward","title":"Exploring Domain Robust Lightweight Reward Models based on Router Mechanism","date":"2024-07-24","arxiv_id":"2407.17546","repositories_listed":0,"syntology":null},{"url":null,"slug":"eegmamba-bidirectional-state-space-models","title":"EEGMamba: Bidirectional State Space Model with Mixture of Experts for EEG Multi-task Classification","date":"2024-07-20","arxiv_id":"2407.20254","repositories_listed":0,"syntology":null},{"url":null,"slug":"evlm-an-efficient-vision-language-model-for","title":"EVLM: An Efficient Vision-Language Model for Visual Understanding","date":"2024-07-19","arxiv_id":"2407.14177","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-of-experts-with-mixture-of-precisions","title":"Mixture of Experts with Mixture of Precisions for Tuning Quality of Service","date":"2024-07-19","arxiv_id":"2407.14417","repositories_listed":0,"syntology":null},{"url":null,"slug":"discussion-effective-and-interpretable","title":"Discussion: Effective and Interpretable Outcome Prediction by Training Sparse Mixtures of Linear Experts","date":"2024-07-18","arxiv_id":"2407.13526","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-of-experts-based-multi-task-supervise","title":"Mixture of Experts based Multi-task Supervise Learning from Crowds","date":"2024-07-18","arxiv_id":"2407.13268","repositories_listed":0,"syntology":null},{"url":null,"slug":"boost-your-nerf-a-model-agnostic-mixture-of","title":"Boost Your NeRF: A Model-Agnostic Mixture of Experts Framework for High Quality and Efficient Rendering","date":"2024-07-15","arxiv_id":"2407.10389","repositories_listed":0,"syntology":null},{"url":"/paper/moe-diffir-task-customized-diffusion-priors","slug":"moe-diffir-task-customized-diffusion-priors","title":"MoE-DiffIR: Task-customized Diffusion Priors for Universal Compressed Image Restoration","date":"2024-07-15","arxiv_id":"2407.10833","repositories_listed":0,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/moe-diffir-task-customized-diffusion-priors#ran","syntology_url":"https://syntology.ai/paper/2407.10833","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.10833"}},"official":null}},{"url":null,"slug":"diversifying-the-expert-knowledge-for-task","title":"Diversifying the Expert Knowledge for Task-Agnostic Pruning in Sparse Mixture-of-Experts","date":"2024-07-12","arxiv_id":"2407.09590","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-unsupervised-domain-adaptation-method-for","title":"An Unsupervised Domain Adaptation Method for Locating Manipulated Region in partially fake Audio","date":"2024-07-11","arxiv_id":"2407.08239","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-simple-architecture-for-enterprise-large","title":"A Simple Architecture for Enterprise Large Language Model Applications based on Role based security and Clearance Levels using Retrieval-Augmented Generation or Mixture of Experts","date":"2024-07-09","arxiv_id":"2407.06718","repositories_listed":0,"syntology":null},{"url":null,"slug":"sam-med3d-moe-towards-a-non-forgetting","title":"SAM-Med3D-MoE: Towards a Non-Forgetting Segment Anything Model via Mixture of Experts for 3D Medical Image Segmentation","date":"2024-07-06","arxiv_id":"2407.04938","repositories_listed":0,"syntology":null},{"url":null,"slug":"lazarus-resilient-and-elastic-training-of","title":"Lazarus: Resilient and Elastic Training of Mixture-of-Experts Models with Adaptive Expert Placement","date":"2024-07-05","arxiv_id":"2407.04656","repositories_listed":0,"syntology":null},{"url":null,"slug":"mobileflow-a-multimodal-llm-for-mobile-gui","title":"MobileFlow: A Multimodal LLM For Mobile GUI Agent","date":"2024-07-05","arxiv_id":"2407.04346","repositories_listed":0,"syntology":null},{"url":null,"slug":"terminating-differentiable-tree-experts","title":"Terminating Differentiable Tree Experts","date":"2024-07-02","arxiv_id":"2407.02060","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-the-potential-of-sparse","title":"Investigating the potential of Sparse Mixtures-of-Experts for multi-domain neural machine translation","date":"2024-07-01","arxiv_id":"2407.01126","repositories_listed":0,"syntology":null},{"url":null,"slug":"sparse-diffusion-policy-a-sparse-reusable-and","title":"Sparse Diffusion Policy: A Sparse, Reusable, and Flexible Policy for Robot Learning","date":"2024-07-01","arxiv_id":"2407.01531","repositories_listed":0,"syntology":null},{"url":null,"slug":"lemoe-advanced-mixture-of-experts-adaptor-for","title":"LEMoE: Advanced Mixture of Experts Adaptor for Lifelong Model Editing of Large Language Models","date":"2024-06-28","arxiv_id":"2406.20030","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-personalized-federated-multi-scenario","title":"Towards Personalized Federated Multi-Scenario Multi-Task Recommendation","date":"2024-06-27","arxiv_id":"2406.18938","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-of-experts-in-a-mixture-of-rl","title":"Mixture of Experts in a Mixture of RL settings","date":"2024-06-26","arxiv_id":"2406.18420","repositories_listed":0,"syntology":null},{"url":null,"slug":"sc-moe-switch-conformer-mixture-of-experts","title":"SC-MoE: Switch Conformer Mixture of Experts for Unified Streaming and Non-streaming Code-Switching ASR","date":"2024-06-26","arxiv_id":"2406.18021","repositories_listed":0,"syntology":null},{"url":null,"slug":"moesd-mixture-of-experts-stable-diffusion-to","title":"MoESD: Mixture of Experts Stable Diffusion to Mitigate Gender Bias","date":"2024-06-25","arxiv_id":"2407.11002","repositories_listed":0,"syntology":null},{"url":null,"slug":"theory-on-mixture-of-experts-in-continual","title":"Theory on Mixture-of-Experts in Continual Learning","date":"2024-06-24","arxiv_id":"2406.16437","repositories_listed":0,"syntology":null},{"url":null,"slug":"simsmoe-solving-representational-collapse-via","title":"SimSMoE: Solving Representational Collapse via Similarity Measure","date":"2024-06-22","arxiv_id":"2406.15883","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-rank-mixture-of-experts-for-continual","title":"Low-Rank Mixture-of-Experts for Continual Medical Image Segmentation","date":"2024-06-19","arxiv_id":"2406.13583","repositories_listed":0,"syntology":null},{"url":null,"slug":"p-tailor-customizing-personality-traits-for","title":"P-Tailor: Customizing Personality Traits for Language Models via Mixture of Specialized LoRA Experts","date":"2024-06-18","arxiv_id":"2406.12548","repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-distillation-of-diffusion","title":"Variational Distillation of Diffusion Policies into Mixture of Experts","date":"2024-06-18","arxiv_id":"2406.12538","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-cascading-mixture-of-experts","title":"Interpretable Cascading Mixture-of-Experts for Urban Traffic Congestion Prediction","date":"2024-06-14","arxiv_id":"2406.12923","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-traffic-forecasting-via-mixture-of","title":"Continual Traffic Forecasting via Mixture of Experts","date":"2024-06-05","arxiv_id":"2406.03140","repositories_listed":0,"syntology":null},{"url":null,"slug":"filtered-not-mixed-stochastic-filtering-based","title":"Filtered not Mixed: Stochastic Filtering-Based Online Gating for Mixture of Large Language Models","date":"2024-06-05","arxiv_id":"2406.02969","repositories_listed":0,"syntology":null},{"url":null,"slug":"node-wise-filtering-in-graph-neural-networks","title":"Node-wise Filtering in Graph Neural Networks: A Mixture of Experts Approach","date":"2024-06-05","arxiv_id":"2406.03464","repositories_listed":0,"syntology":null},{"url":null,"slug":"style-mixture-of-experts-for-expressive-text","title":"Style Mixture of Experts for Expressive Text-To-Speech Synthesis","date":"2024-06-05","arxiv_id":"2406.03637","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-6g-integrated-sensing-and","title":"Optimizing 6G Integrated Sensing and Communications (ISAC) via Expert Networks","date":"2024-06-01","arxiv_id":"2406.00408","repositories_listed":0,"syntology":null},{"url":null,"slug":"training-efficient-density-quantum-machine","title":"Training-efficient density quantum machine learning","date":"2024-05-30","arxiv_id":"2405.20237","repositories_listed":0,"syntology":null},{"url":null,"slug":"memoe-enhancing-model-editing-with-mixture-of","title":"MEMoE: Enhancing Model Editing with Mixture of Experts Adaptors","date":"2024-05-29","arxiv_id":"2405.19086","repositories_listed":0,"syntology":null},{"url":null,"slug":"monde-mixture-of-near-data-experts-for-large","title":"MoNDE: Mixture of Near-Data Experts for Large-Scale Sparse Models","date":"2024-05-29","arxiv_id":"2405.18832","repositories_listed":0,"syntology":null},{"url":null,"slug":"lora-switch-boosting-the-efficiency-of","title":"LoRA-Switch: Boosting the Efficiency of Dynamic LLM Adapters via System-Algorithm Co-design","date":"2024-05-28","arxiv_id":"2405.17741","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-provably-effective-method-for-pruning","title":"A Provably Effective Method for Pruning Experts in Fine-tuned Sparse Mixture-of-Experts","date":"2024-05-26","arxiv_id":"2405.16646","repositories_listed":0,"syntology":null},{"url":null,"slug":"expert-token-resonance-redefining-moe-routing","title":"Expert-Token Resonance: Redefining MoE Routing through Affinity-Driven Active Selection","date":"2024-05-24","arxiv_id":"2406.00023","repositories_listed":0,"syntology":null},{"url":null,"slug":"statistical-advantages-of-perturbing-cosine","title":"Statistical Advantages of Perturbing Cosine Router in Mixture of Experts","date":"2024-05-23","arxiv_id":"2405.14131","repositories_listed":0,"syntology":null},{"url":null,"slug":"sigmoid-gating-is-more-sample-efficient-than","title":"Sigmoid Gating is More Sample Efficient than Softmax Gating in Mixture of Experts","date":"2024-05-22","arxiv_id":"2405.13997","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-more-generalized-experts-by-merging","title":"Learning More Generalized Experts by Merging Experts in Mixture-of-Experts","date":"2024-05-19","arxiv_id":"2405.11530","repositories_listed":0,"syntology":null},{"url":null,"slug":"many-hands-make-light-work-task-oriented","title":"Many Hands Make Light Work: Task-Oriented Dialogue System with Module-Based Mixture-of-Experts","date":"2024-05-16","arxiv_id":"2405.09744","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-mixture-of-experts-approach-to-few-shot","title":"A Mixture-of-Experts Approach to Few-Shot Task Transfer in Open-Ended Text Worlds","date":"2024-05-09","arxiv_id":"2405.06059","repositories_listed":0,"syntology":null},{"url":null,"slug":"sutra-scalable-multilingual-language-model","title":"SUTRA: Scalable Multilingual Language Model Architecture","date":"2024-05-07","arxiv_id":"2405.06694","repositories_listed":0,"syntology":null},{"url":null,"slug":"lory-fully-differentiable-mixture-of-experts","title":"Lory: Fully Differentiable Mixture-of-Experts for Autoregressive Language Model Pre-training","date":"2024-05-06","arxiv_id":"2405.03133","repositories_listed":0,"syntology":null},{"url":null,"slug":"meet-mixture-of-experts-extra-tree-based-semg","title":"MEET: Mixture of Experts Extra Tree-Based sEMG Hand Gesture Identification","date":"2024-05-06","arxiv_id":"2405.09562","repositories_listed":0,"syntology":null},{"url":null,"slug":"wdmoe-wireless-distributed-large-language","title":"WDMoE: Wireless Distributed Large Language Models with Mixture of Experts","date":"2024-05-06","arxiv_id":"2405.03131","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-of-partially-linear-experts","title":"Mixture of partially linear experts","date":"2024-05-05","arxiv_id":"2405.02905","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-mixture-of-discriminative","title":"Hierarchical mixture of discriminative Generalized Dirichlet classifiers","date":"2024-05-02","arxiv_id":"2405.01778","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-of-insightful-experts-mote-the","title":"Mixture of insighTful Experts (MoTE): The Synergy of Thought Chains and Expert Mixtures in Self-Alignment","date":"2024-05-01","arxiv_id":"2405.00557","repositories_listed":0,"syntology":null},{"url":null,"slug":"mopeft-a-mixture-of-pefts-for-the-segment","title":"MoPEFT: A Mixture-of-PEFTs for the Segment Anything Model","date":"2024-05-01","arxiv_id":"2405.00293","repositories_listed":0,"syntology":null},{"url":null,"slug":"powering-in-database-dynamic-model-slicing","title":"Powering In-Database Dynamic Model Slicing for Structured Data Analytics","date":"2024-05-01","arxiv_id":"2405.00568","repositories_listed":0,"syntology":null},{"url":null,"slug":"lancet-accelerating-mixture-of-experts","title":"Lancet: Accelerating Mixture-of-Experts Training via Whole Graph Computation-Communication Overlapping","date":"2024-04-30","arxiv_id":"2404.19429","repositories_listed":0,"syntology":null},{"url":null,"slug":"mix-of-experts-language-model-for-named","title":"Mix of Experts Language Model for Named Entity Recognition","date":"2024-04-30","arxiv_id":"2404.19192","repositories_listed":0,"syntology":null},{"url":null,"slug":"trends-and-challenges-of-real-time-learning","title":"Towards Incremental Learning in Large Language Models: A Critical Review","date":"2024-04-28","arxiv_id":"2404.18311","repositories_listed":0,"syntology":null},{"url":null,"slug":"integration-of-mixture-of-experts-and","title":"Integration of Mixture of Experts and Multimodal Generative AI in Internet of Vehicles: A Survey","date":"2024-04-25","arxiv_id":"2404.16356","repositories_listed":0,"syntology":null},{"url":null,"slug":"u2-moe-scaling-4-7x-parameters-with-minimal","title":"U2++ MoE: Scaling 4.7x parameters with minimal impact on RTF","date":"2024-04-25","arxiv_id":"2404.16407","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-a-i-enhanced-reservoir","title":"A Novel A.I Enhanced Reservoir Characterization with a Combined Mixture of Experts -- NVIDIA Modulus based Physics Informed Neural Operator Forward Model","date":"2024-04-20","arxiv_id":"2404.14447","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-large-scale-medical-visual-task-adaptation","title":"A Large-scale Medical Visual Task Adaptation Benchmark","date":"2024-04-19","arxiv_id":"2404.12876","repositories_listed":0,"syntology":null},{"url":null,"slug":"moa-mixture-of-attention-for-subject-context","title":"MoA: Mixture-of-Attention for Subject-Context Disentanglement in Personalized Image Generation","date":"2024-04-17","arxiv_id":"2404.11565","repositories_listed":0,"syntology":null},{"url":null,"slug":"interactive-generative-ai-agents-for","title":"Generative AI Agents with Large Language Model for Satellite Networks via a Mixture of Experts Transmission","date":"2024-04-14","arxiv_id":"2404.09134","repositories_listed":0,"syntology":null},{"url":null,"slug":"intuition-aware-mixture-of-rank-1-experts-for","title":"Intuition-aware Mixture-of-Rank-1-Experts for Parameter Efficient Finetuning","date":"2024-04-13","arxiv_id":"2404.08985","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-of-experts-soften-the-curse-of","title":"Mixture of Experts Soften the Curse of Dimensionality in Operator Learning","date":"2024-04-13","arxiv_id":"2404.09101","repositories_listed":0,"syntology":null},{"url":null,"slug":"identifying-shopping-intent-in-product-qa-for","title":"Identifying Shopping Intent in Product QA for Proactive Recommendations","date":"2024-04-09","arxiv_id":"2404.06017","repositories_listed":0,"syntology":null},{"url":null,"slug":"dense-training-sparse-inference-rethinking","title":"Dense Training, Sparse Inference: Rethinking Training of Mixture-of-Experts Language Models","date":"2024-04-08","arxiv_id":"2404.05567","repositories_listed":0,"syntology":null},{"url":null,"slug":"seer-moe-sparse-expert-efficiency-through","title":"SEER-MoE: Sparse Expert Efficiency through Regularization for Mixture-of-Experts","date":"2024-04-07","arxiv_id":"2404.05089","repositories_listed":0,"syntology":null},{"url":null,"slug":"shortcut-connected-expert-parallelism-for","title":"Shortcut-connected Expert Parallelism for Accelerating Mixture-of-Experts","date":"2024-04-07","arxiv_id":"2404.05019","repositories_listed":0,"syntology":null}],"record_sha256":"83aacca57bb1223d923395585675b6a8d3e9dbb9b31918708e5fc8d3594d685e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}