{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/mixture-of-experts/papers/8","list_of":"/task/mixture-of-experts","task":"Mixture-of-Experts","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":8,"pages_in_order":14,"rows_per_page":100,"rows":[701,800],"of":1312,"counts":{"archive_papers_tagged":1312,"with_a_code_link":516,"where_syntology_ran_a_sample":216,"not_listed_spam_title":0,"listed":1312,"listed_where_code_ran":216,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":184,"every_run_a_failure_of_syntologys_instrument":32,"listed_with_a_run_with_no_instrument_failure":184,"listed_every_run_a_failure_of_syntologys_instrument":32,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/mixture-of-experts","prev":"/task/mixture-of-experts/papers/7","next":"/task/mixture-of-experts/papers/9","papers":[{"url":null,"slug":"speculative-moe-communication-efficient","title":"Speculative MoE: Communication Efficient Parallel MoE Inference with Speculative Token and Expert Pre-scheduling","date":"2025-03-06","arxiv_id":"2503.04398","repositories_listed":0,"syntology":null},{"url":null,"slug":"ts-rag-retrieval-augmented-generation-based","title":"TS-RAG: Retrieval-Augmented Generation based Time Series Foundation Models are Stronger Zero-Shot Forecaster","date":"2025-03-06","arxiv_id":"2503.07649","repositories_listed":0,"syntology":null},{"url":null,"slug":"brainnet-moe-brain-inspired-mixture-of","title":"BrainNet-MoE: Brain-Inspired Mixture-of-Experts Learning for Neurological Disease Identification","date":"2025-03-05","arxiv_id":"2503.07640","repositories_listed":0,"syntology":null},{"url":null,"slug":"convergence-rates-for-softmax-gating-mixture","title":"Convergence Rates for Softmax Gating Mixture of Experts","date":"2025-03-05","arxiv_id":"2503.03213","repositories_listed":0,"syntology":null},{"url":null,"slug":"tabby-tabular-data-synthesis-with-language","title":"Tabby: Tabular Data Synthesis with Language Models","date":"2025-03-04","arxiv_id":"2503.02152","repositories_listed":0,"syntology":null},{"url":null,"slug":"ders-towards-extremely-efficient-upcycled","title":"DeRS: Towards Extremely Efficient Upcycled Mixture-of-Experts Models","date":"2025-03-03","arxiv_id":"2503.01359","repositories_listed":0,"syntology":null},{"url":null,"slug":"ecg-emotionnet-nested-mixture-of-expert-nmoe","title":"ECG-EmotionNet: Nested Mixture of Expert (NMoE) Adaptation of ECG-Foundation Model for Driver Emotion Recognition","date":"2025-03-03","arxiv_id":"2503.01750","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-do-consumers-really-choose-exposing","title":"How Do Consumers Really Choose: Exposing Hidden Preferences with the Mixture of Experts Model","date":"2025-03-03","arxiv_id":"2503.05800","repositories_listed":0,"syntology":null},{"url":"/paper/proper-a-progressive-learning-framework-for","slug":"proper-a-progressive-learning-framework-for","title":"PROPER: A Progressive Learning Framework for Personalized Large Language Models with Group-Level Adaptation","date":"2025-03-03","arxiv_id":"2503.01303","repositories_listed":0,"syntology":{"n":5,"n_ran":3,"n_constructed":2,"n_ran_checked":2,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/proper-a-progressive-learning-framework-for#ran","syntology_url":"https://syntology.ai/paper/2503.01303","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.01303"}},"official":null}},{"url":null,"slug":"unify-and-anchor-a-context-aware-transformer","title":"Unify and Anchor: A Context-Aware Transformer for Cross-Domain Time Series Forecasting","date":"2025-03-03","arxiv_id":"2503.01157","repositories_listed":0,"syntology":null},{"url":null,"slug":"explainable-classifier-for-malignant-lymphoma","title":"Explainable Classifier for Malignant Lymphoma Subtyping via Cell Graph and Image Fusion","date":"2025-03-02","arxiv_id":"2503.00925","repositories_listed":0,"syntology":null},{"url":null,"slug":"cl-moe-enhancing-multimodal-large-language","title":"CL-MoE: Enhancing Multimodal Large Language Model with Dual Momentum Mixture-of-Experts for Continual Visual Question Answering","date":"2025-03-01","arxiv_id":"2503.00413","repositories_listed":0,"syntology":null},{"url":null,"slug":"cosmoes-compact-sparse-mixture-of-experts","title":"CoSMoEs: Compact Sparse Mixture of Experts","date":"2025-02-28","arxiv_id":"2503.00245","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-of-experts-augmented-deep-unfolding","title":"Mixture of Experts-augmented Deep Unfolding for Activity Detection in IRS-aided Systems","date":"2025-02-27","arxiv_id":"2502.20183","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-of-experts-for-recognizing-depression","title":"Mixture of Experts for Recognizing Depression from Interview and Reading Tasks","date":"2025-02-27","arxiv_id":"2502.20213","repositories_listed":0,"syntology":null},{"url":null,"slug":"unicodec-unified-audio-codec-with-single","title":"UniCodec: Unified Audio Codec with Single Domain-Adaptive Codebook","date":"2025-02-27","arxiv_id":"2502.20067","repositories_listed":0,"syntology":null},{"url":"/paper/drop-upcycling-training-sparse-mixture-of","slug":"drop-upcycling-training-sparse-mixture-of","title":"Drop-Upcycling: Training Sparse Mixture of Experts with Partial Re-initialization","date":"2025-02-26","arxiv_id":"2502.19261","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/drop-upcycling-training-sparse-mixture-of#ran","syntology_url":"https://syntology.ai/paper/2502.19261","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.19261"}},"official":null}},{"url":null,"slug":"onerec-unifying-retrieve-and-rank-with","title":"OneRec: Unifying Retrieve and Rank with Generative Recommender and Iterative Preference Alignment","date":"2025-02-26","arxiv_id":"2502.18965","repositories_listed":0,"syntology":null},{"url":null,"slug":"bigmac-a-communication-efficient-mixture-of","title":"BigMac: A Communication-Efficient Mixture-of-Experts Model Structure for Fast Training and Inference","date":"2025-02-24","arxiv_id":"2502.16927","repositories_listed":0,"syntology":null},{"url":null,"slug":"enact-heart-ensemble-based-assessment-using","title":"ENACT-Heart -- ENsemble-based Assessment Using CNN and Transformer on Heart Sounds","date":"2025-02-24","arxiv_id":"2502.16914","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-expert-contributions-in-a-moe-llm","title":"Evaluating Expert Contributions in a MoE LLM for Quiz-Based Tasks","date":"2025-02-24","arxiv_id":"2502.17187","repositories_listed":0,"syntology":null},{"url":"/paper/the-empirical-impact-of-reducing-symmetries","slug":"the-empirical-impact-of-reducing-symmetries","title":"The Empirical Impact of Reducing Symmetries on the Performance of Deep Ensembles and MoE","date":"2025-02-24","arxiv_id":"2502.17391","repositories_listed":0,"syntology":{"n":7,"n_ran":6,"n_constructed":3,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":3,"n_violates":0,"n_no_contract":3,"n_pointer_only":7,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 6 with no instrument failure: 3 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-empirical-impact-of-reducing-symmetries#ran","syntology_url":"https://syntology.ai/paper/2502.17391","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.17391"}},"official":null}},{"url":null,"slug":"an-autonomous-network-orchestration-framework","title":"An Autonomous Network Orchestration Framework Integrating Large Language Models with Continual Reinforcement Learning","date":"2025-02-22","arxiv_id":"2502.16198","repositories_listed":0,"syntology":null},{"url":null,"slug":"ray-tracing-for-conditionally-activated","title":"Ray-Tracing for Conditionally Activated Neural Networks","date":"2025-02-20","arxiv_id":"2502.14788","repositories_listed":0,"syntology":null},{"url":null,"slug":"unraveling-the-localized-latents-learning","title":"Unraveling the Localized Latents: Learning Stratified Manifold Structures in LLM Embedding Space with Sparse Mixture-of-Experts","date":"2025-02-19","arxiv_id":"2502.13577","repositories_listed":0,"syntology":null},{"url":null,"slug":"dsmoe-matrix-partitioned-experts-with-dynamic","title":"DSMoE: Matrix-Partitioned Experts with Dynamic Routing for Computation-Efficient Dense LLMs","date":"2025-02-18","arxiv_id":"2502.12455","repositories_listed":0,"syntology":null},{"url":null,"slug":"every-expert-matters-towards-effective","title":"Every Expert Matters: Towards Effective Knowledge Distillation for Mixture-of-Experts Language Models","date":"2025-02-18","arxiv_id":"2502.12947","repositories_listed":0,"syntology":null},{"url":null,"slug":"connector-s-a-survey-of-connectors-in-multi","title":"Connector-S: A Survey of Connectors in Multi-modal Large Language Models","date":"2025-02-17","arxiv_id":"2502.11453","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-to-upscale-neural-networks-with-scaling","title":"How to Upscale Neural Networks with Scaling Law? A Survey and Practical Guidelines","date":"2025-02-17","arxiv_id":"2502.12051","repositories_listed":0,"syntology":null},{"url":null,"slug":"climatellm-efficient-weather-forecasting-via","title":"ClimateLLM: Efficient Weather Forecasting via Frequency-Aware Large Language Models","date":"2025-02-16","arxiv_id":"2502.11059","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-of-tunable-experts-behavior","title":"Mixture of Tunable Experts - Behavior Modification of DeepSeek-R1 at Inference Time","date":"2025-02-16","arxiv_id":"2502.11096","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-specialization-in-moe-appears-with","title":"Probing Semantic Routing in Large Mixture-of-Expert Models","date":"2025-02-15","arxiv_id":"2502.10928","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-of-decoupled-message-passing-experts","title":"Mixture of Decoupled Message Passing Experts with Entropy Constraint for General Node Classification","date":"2025-02-12","arxiv_id":"2502.08083","repositories_listed":0,"syntology":null},{"url":null,"slug":"memory-analysis-on-the-training-course-of","title":"Memory Analysis on the Training Course of DeepSeek Models","date":"2025-02-11","arxiv_id":"2502.07846","repositories_listed":0,"syntology":null},{"url":null,"slug":"moenas-mixture-of-expert-based-neural","title":"MoENAS: Mixture-of-Expert based Neural Architecture Search for jointly Accurate, Fair, and Robust Edge Deep Neural Networks","date":"2025-02-11","arxiv_id":"2502.07422","repositories_listed":0,"syntology":null},{"url":null,"slug":"mohave-mixture-of-hierarchical-audio-visual","title":"MoHAVE: Mixture of Hierarchical Audio-Visual Experts for Robust Speech Recognition","date":"2025-02-11","arxiv_id":"2502.10447","repositories_listed":0,"syntology":null},{"url":null,"slug":"moetuner-optimized-mixture-of-expert-serving","title":"MoETuner: Optimized Mixture of Expert Serving with Balanced Expert Placement and Token Routing","date":"2025-02-10","arxiv_id":"2502.06643","repositories_listed":0,"syntology":null},{"url":null,"slug":"moemba-a-mamba-based-mixture-of-experts-for","title":"MoEMba: A Mamba-based Mixture of Experts for High-Density EMG-based Hand Gesture Recognition","date":"2025-02-09","arxiv_id":"2502.17457","repositories_listed":0,"syntology":null},{"url":null,"slug":"fmoe-fine-grained-expert-offloading-for-large","title":"fMoE: Fine-Grained Expert Offloading for Large Mixture-of-Experts Serving","date":"2025-02-07","arxiv_id":"2502.05370","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-moe-scaling-laws-mixture-of-experts-can","title":"Joint MoE Scaling Laws: Mixture of Experts Can Be Memory Efficient","date":"2025-02-07","arxiv_id":"2502.05172","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-pre-trained-models-for-multimodal","title":"Leveraging Pre-Trained Models for Multimodal Class-Incremental Learning under Adaptive Fusion","date":"2025-02-07","arxiv_id":"2506.09999","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-foundational-models-for-dynamical","title":"Towards Foundational Models for Dynamical System Reconstruction: Hierarchical Meta-Learning via Mixture of Experts","date":"2025-02-07","arxiv_id":"2502.05335","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-of-neural-operator-experts-for","title":"Mixture of neural operator experts for learning boundary conditions and model selection","date":"2025-02-06","arxiv_id":"2502.04562","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-robustness-and-accuracy-in-mixture","title":"Optimizing Robustness and Accuracy in Mixture of Experts: A Dual-Model Approach","date":"2025-02-05","arxiv_id":"2502.06832","repositories_listed":0,"syntology":null},{"url":null,"slug":"brief-analysis-of-deepseek-r1-and-it-s","title":"Brief analysis of DeepSeek R1 and it's implications for Generative AI","date":"2025-02-04","arxiv_id":"2502.02523","repositories_listed":0,"syntology":null},{"url":null,"slug":"m2r2-mixture-of-multi-rate-residuals-for","title":"M2R2: Mixture of Multi-Rate Residuals for Efficient Transformer Inference","date":"2025-02-04","arxiv_id":"2502.02040","repositories_listed":0,"syntology":null},{"url":null,"slug":"regnet-reciprocal-space-aware-long-range","title":"ReGNet: Reciprocal Space-Aware Long-Range Modeling for Crystalline Property Prediction","date":"2025-02-04","arxiv_id":"2502.02748","repositories_listed":0,"syntology":null},{"url":null,"slug":"clip-up-a-simple-and-efficient-mixture-of","title":"CLIP-UP: A Simple and Efficient Mixture-of-Experts CLIP Training Recipe with Sparse Upcycling","date":"2025-02-03","arxiv_id":"2502.00965","repositories_listed":0,"syntology":null},{"url":null,"slug":"mergeme-model-merging-techniques-for","title":"MergeME: Model Merging Techniques for Homogeneous and Heterogeneous MoEs","date":"2025-02-03","arxiv_id":"2502.00997","repositories_listed":0,"syntology":null},{"url":null,"slug":"mj-video-fine-grained-benchmarking-and","title":"MJ-VIDEO: Fine-Grained Benchmarking and Rewarding Video Preferences in Video Generation","date":"2025-02-03","arxiv_id":"2502.01719","repositories_listed":0,"syntology":null},{"url":null,"slug":"sigmoid-self-attention-has-lower-sample","title":"Sigmoid Self-Attention has Lower Sample Complexity than Softmax Self-Attention: A Mixture-of-Experts Perspective","date":"2025-02-01","arxiv_id":"2502.00281","repositories_listed":0,"syntology":null},{"url":"/paper/adaptive-prompt-unlocking-the-power-of-visual","slug":"adaptive-prompt-unlocking-the-power-of-visual","title":"Adaptive Prompt: Unlocking the Power of Visual Prompt Tuning","date":"2025-01-31","arxiv_id":"2501.18936","repositories_listed":0,"syntology":{"n":12,"n_ran":7,"n_constructed":6,"n_ran_checked":7,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":12,"phrase":"7 ran (of which 6 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/adaptive-prompt-unlocking-the-power-of-visual#ran","syntology_url":"https://syntology.ai/paper/2501.18936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.18936"}},"official":null}},{"url":null,"slug":"pheromone-based-learning-of-optimal-reasoning","title":"Pheromone-based Learning of Optimal Reasoning Paths","date":"2025-01-31","arxiv_id":"2501.19278","repositories_listed":0,"syntology":null},{"url":null,"slug":"molgraph-xlstm-a-graph-based-dual-level-xlstm","title":"MolGraph-xLSTM: A graph-based dual-level xLSTM framework with multi-head mixture-of-experts for enhanced molecular representation and interpretability","date":"2025-01-30","arxiv_id":"2501.18439","repositories_listed":0,"syntology":null},{"url":null,"slug":"free-agent-in-agent-based-mixture-of-experts","title":"Free Agent in Agent-Based Mixture-of-Experts Generative AI Framework","date":"2025-01-29","arxiv_id":"2501.17903","repositories_listed":0,"syntology":null},{"url":null,"slug":"heuristic-informed-mixture-of-experts-for","title":"Heuristic-Informed Mixture of Experts for Link Prediction in Multilayer Networks","date":"2025-01-29","arxiv_id":"2501.17557","repositories_listed":0,"syntology":null},{"url":null,"slug":"3d-moe-a-mixture-of-experts-multi-modal-llm","title":"3D-MoE: A Mixture-of-Experts Multi-modal LLM for 3D Vision and Pose Diffusion via Rectified Flow","date":"2025-01-28","arxiv_id":"2501.16698","repositories_listed":0,"syntology":null},{"url":null,"slug":"static-batching-of-irregular-workloads-on","title":"Static Batching of Irregular Workloads on GPUs: Framework and Application to Efficient MoE Model Inference","date":"2025-01-27","arxiv_id":"2501.16103","repositories_listed":0,"syntology":null},{"url":null,"slug":"each-rank-could-be-an-expert-single-ranked","title":"Each Rank Could be an Expert: Single-Ranked Mixture of Experts LoRA for Multi-Task Learning","date":"2025-01-25","arxiv_id":"2501.15103","repositories_listed":0,"syntology":null},{"url":null,"slug":"tomoe-converting-dense-large-language-models","title":"ToMoE: Converting Dense Large Language Models to Mixture-of-Experts through Dynamic Structural Pruning","date":"2025-01-25","arxiv_id":"2501.15316","repositories_listed":0,"syntology":null},{"url":null,"slug":"mean-field-limit-from-general-mixtures-of","title":"Mean-field limit from general mixtures of experts to quantum neural networks","date":"2025-01-24","arxiv_id":"2501.14660","repositories_listed":0,"syntology":null},{"url":null,"slug":"sparse-mixture-of-experts-for-non-uniform","title":"Sparse Mixture-of-Experts for Non-Uniform Noise Reduction in MRI Images","date":"2025-01-24","arxiv_id":"2501.14198","repositories_listed":0,"syntology":null},{"url":null,"slug":"csaot-cooperative-multi-agent-system-for","title":"CSAOT: Cooperative Multi-Agent System for Active Object Tracking","date":"2025-01-23","arxiv_id":"2501.13994","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomy-of-experts-models","title":"Autonomy-of-Experts Models","date":"2025-01-22","arxiv_id":"2501.13074","repositories_listed":0,"syntology":null},{"url":null,"slug":"blr-moe-boosted-language-routing-mixture-of","title":"BLR-MoE: Boosted Language-Routing Mixture of Experts for Domain-Robust Multilingual E2E ASR","date":"2025-01-22","arxiv_id":"2501.12602","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm4wm-adapting-llm-for-wireless-multi","title":"LLM4WM: Adapting LLM for Wireless Multi-Tasking","date":"2025-01-22","arxiv_id":"2501.12983","repositories_listed":0,"syntology":null},{"url":null,"slug":"uniuir-considering-underwater-image","title":"UniUIR: Considering Underwater Image Restoration as An All-in-One Learner","date":"2025-01-22","arxiv_id":"2501.12981","repositories_listed":0,"syntology":null},{"url":null,"slug":"demons-in-the-detail-on-implementing-load","title":"Demons in the Detail: On Implementing Load Balancing Loss for Training Specialized Mixture-of-Expert Models","date":"2025-01-21","arxiv_id":"2501.11873","repositories_listed":0,"syntology":null},{"url":null,"slug":"parameters-vs-flops-scaling-laws-for-optimal","title":"Parameters vs FLOPs: Scaling Laws for Optimal Sparsity for Mixture-of-Experts Language Models","date":"2025-01-21","arxiv_id":"2501.12370","repositories_listed":0,"syntology":null},{"url":null,"slug":"scfcrc-simultaneously-counteract-feature","title":"SCFCRC: Simultaneously Counteract Feature Camouflage and Relation Camouflage for Fraud Detection","date":"2025-01-21","arxiv_id":"2501.12430","repositories_listed":0,"syntology":null},{"url":null,"slug":"fsmoe-a-flexible-and-scalable-training-system","title":"FSMoE: A Flexible and Scalable Training System for Sparse Mixture-of-Experts Models","date":"2025-01-18","arxiv_id":"2501.10714","repositories_listed":0,"syntology":null},{"url":null,"slug":"omoe-diversifying-mixture-of-low-rank","title":"OMoE: Diversifying Mixture of Low-Rank Adaptation by Orthogonal Finetuning","date":"2025-01-17","arxiv_id":"2501.10062","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-based-routing-in-mixture-of-experts-a","title":"LLM-Based Routing in Mixture of Experts: A Novel Framework for Trading","date":"2025-01-16","arxiv_id":"2501.09636","repositories_listed":0,"syntology":null},{"url":null,"slug":"graphmoe-amplifying-cognitive-depth-of","title":"GRAPHMOE: Amplifying Cognitive Depth of Mixture-of-Experts Network via Introducing Self-Rethinking Mechanism","date":"2025-01-14","arxiv_id":"2501.07890","repositories_listed":0,"syntology":null},{"url":null,"slug":"psreg-prior-guided-sparse-mixture-of-experts","title":"PSReg: Prior-guided Sparse Mixture of Experts for Point Cloud Registration","date":"2025-01-14","arxiv_id":"2501.07762","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-distributed-deployment-of-mixture","title":"Optimizing Distributed Deployment of Mixture-of-Experts Model Inference in Serverless Computing","date":"2025-01-09","arxiv_id":"2501.05313","repositories_listed":0,"syntology":null},{"url":null,"slug":"mfabric-an-efficient-and-scalable-fabric-for","title":"mFabric: An Efficient and Scalable Fabric for Mixture-of-Experts Training","date":"2025-01-07","arxiv_id":"2501.03905","repositories_listed":0,"syntology":null},{"url":null,"slug":"fresh-cl-feature-realignment-through-experts","title":"Fresh-CL: Feature Realignment through Experts on Hypersphere in Continual Learning","date":"2025-01-04","arxiv_id":"2501.02198","repositories_listed":0,"syntology":null},{"url":"/paper/move-kd-knowledge-distillation-for-vlms-with","slug":"move-kd-knowledge-distillation-for-vlms-with","title":"MoVE-KD: Knowledge Distillation for VLMs with Mixture of Visual Encoders","date":"2025-01-03","arxiv_id":"2501.01709","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/move-kd-knowledge-distillation-for-vlms-with#ran","syntology_url":"https://syntology.ai/paper/2501.01709","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.01709"}},"official":null}},{"url":null,"slug":"correlative-and-discriminative-label-grouping","title":"Correlative and Discriminative Label Grouping for Multi-Label Visual Prompt Tuning","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-heterogeneous-tissues-with-mixture","title":"Learning Heterogeneous Tissues with Mixture of Experts for Gigapixel Whole Slide Images","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"mexd-an-expert-infused-diffusion-model-for","title":"MExD: An Expert-Infused Diffusion Model for Whole-Slide Image Classification","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"rem-a-scalable-reinforced-multi-expert","title":"REM: A Scalable Reinforced Multi-Expert Framework for Multiplex Influence Maximization","date":"2025-01-01","arxiv_id":"2501.00779","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-efficient-foundation-model-for-zero","title":"Towards Efficient Foundation Model for Zero-shot Amodal Segmentation","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"unialign-scaling-multimodal-alignment-within","title":"UNIALIGN: Scaling Multimodal Alignment within One Unified Model","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"cnc-cross-modal-normality-constraint-for","title":"CNC: Cross-modal Normality Constraint for Unsupervised Multi-class Anomaly Detection","date":"2024-12-31","arxiv_id":"2501.00346","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-variational-autoencoder-a","title":"Multimodal Variational Autoencoder: a Barycentric View","date":"2024-12-29","arxiv_id":"2412.20487","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-mixture-of-experts-and-memory-augmented","title":"Graph Mixture of Experts and Memory-augmented Routers for Multivariate Time Series Anomaly Detection","date":"2024-12-26","arxiv_id":"2412.19108","repositories_listed":0,"syntology":null},{"url":null,"slug":"ume-upcycling-mixture-of-experts-for-scalable","title":"UME: Upcycling Mixture-of-Experts for Scalable and Efficient Automatic Speech Recognition","date":"2024-12-23","arxiv_id":"2412.17507","repositories_listed":0,"syntology":null},{"url":null,"slug":"part-of-speech-sensitivity-of-routers-in","title":"Part-Of-Speech Sensitivity of Routers in Mixture of Experts Models","date":"2024-12-22","arxiv_id":"2412.16971","repositories_listed":0,"syntology":null},{"url":null,"slug":"theory-of-mixture-of-experts-for-mobile-edge","title":"Theory of Mixture-of-Experts for Mobile Edge Computing","date":"2024-12-20","arxiv_id":"2412.15690","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-healthcare-recommendation-systems","title":"Enhancing Healthcare Recommendation Systems with a Multimodal LLMs-based MOE Architecture","date":"2024-12-16","arxiv_id":"2412.11557","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-mixture-of-experts-in-dense","title":"Investigating Mixture of Experts in Dense Retrieval","date":"2024-12-16","arxiv_id":"2412.11864","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-prompting-for-continual-relation","title":"Adaptive Prompting for Continual Relation Extraction: A Within-Task Variance Perspective","date":"2024-12-11","arxiv_id":"2412.08285","repositories_listed":0,"syntology":null},{"url":null,"slug":"moe-cap-cost-accuracy-performance","title":"MoE-CAP: Benchmarking Cost, Accuracy and Performance of Sparse Mixture-of-Experts Systems","date":"2024-12-10","arxiv_id":"2412.07067","repositories_listed":0,"syntology":null},{"url":null,"slug":"unipaint-unified-space-time-video-inpainting","title":"UniPaint: Unified Space-time Video Inpainting via Mixture-of-Experts","date":"2024-12-09","arxiv_id":"2412.06340","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-entailment-tree-generation-approach-for","title":"An Entailment Tree Generation Approach for Multimodal Multi-Hop Question Answering with Mixture-of-Experts and Iterative Feedback Mechanism","date":"2024-12-08","arxiv_id":"2412.05821","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-3d-acceleration-for-low-power-mixture","title":"Towards 3D Acceleration for low-power Mixture-of-Experts and Multi-Head Attention Spiking Transformers","date":"2024-12-07","arxiv_id":"2412.05540","repositories_listed":0,"syntology":null},{"url":null,"slug":"steps-are-all-you-need-rethinking-stem","title":"Steps are all you need: Rethinking STEM Education with Prompt Engineering","date":"2024-12-06","arxiv_id":"2412.05023","repositories_listed":0,"syntology":null},{"url":null,"slug":"convolutional-neural-networks-and-mixture-of","title":"Convolutional Neural Networks and Mixture of Experts for Intrusion Detection in 5G Networks and beyond","date":"2024-12-04","arxiv_id":"2412.03483","repositories_listed":0,"syntology":null}],"record_sha256":"6c5a4e6a16c56bcdd3c0eac481ce377095e3166a7dc09805e7330d97d1c3a66f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}