{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/mixture-of-experts/papers/6","list_of":"/task/mixture-of-experts","task":"Mixture-of-Experts","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":6,"pages_in_order":14,"rows_per_page":100,"rows":[501,600],"of":1312,"counts":{"archive_papers_tagged":1312,"with_a_code_link":516,"where_syntology_ran_a_sample":216,"not_listed_spam_title":0,"listed":1312,"listed_where_code_ran":216,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":184,"every_run_a_failure_of_syntologys_instrument":32,"listed_with_a_run_with_no_instrument_failure":184,"listed_every_run_a_failure_of_syntologys_instrument":32,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/mixture-of-experts","prev":"/task/mixture-of-experts/papers/5","next":"/task/mixture-of-experts/papers/7","papers":[{"url":"/paper/a-modular-task-oriented-dialogue-system-using","slug":"a-modular-task-oriented-dialogue-system-using","title":"A Modular Task-oriented Dialogue System Using a Neural Mixture-of-Experts","date":"2019-07-10","arxiv_id":"1907.05346","repositories_listed":1,"syntology":null},{"url":"/paper/sequential-gaussian-processes-for-online","slug":"sequential-gaussian-processes-for-online","title":"Sequential Gaussian Processes for Online Learning of Nonstationary Functions","date":"2019-05-24","arxiv_id":"1905.10003","repositories_listed":1,"syntology":null},{"url":"/paper/tensor-variate-mixture-of-experts","slug":"tensor-variate-mixture-of-experts","title":"Tensor-variate Mixture of Experts for Proportional Myographic Control of a Robotic Hand","date":"2019-02-28","arxiv_id":"1902.11104","repositories_listed":1,"syntology":null},{"url":"/paper/nesti-net-normal-estimation-for-unstructured","slug":"nesti-net-normal-estimation-for-unstructured","title":"Nesti-Net: Normal Estimation for Unstructured 3D Point Clouds using Convolutional Neural Networks","date":"2018-12-03","arxiv_id":"1812.00709","repositories_listed":1,"syntology":null},{"url":"/paper/zero-resource-multilingual-model-transfer","slug":"zero-resource-multilingual-model-transfer","title":"Multi-Source Cross-Lingual Model Transfer: Learning What to Share","date":"2018-10-08","arxiv_id":"1810.03552","repositories_listed":1,"syntology":null},{"url":"/paper/learning-deep-mixtures-of-gaussian-process","slug":"learning-deep-mixtures-of-gaussian-process","title":"Learning Deep Mixtures of Gaussian Process Experts Using Sum-Product Networks","date":"2018-09-12","arxiv_id":"1809.04400","repositories_listed":1,"syntology":null},{"url":"/paper/multi-source-domain-adaptation-with-mixture","slug":"multi-source-domain-adaptation-with-mixture","title":"Multi-Source Domain Adaptation with Mixture of Experts","date":"2018-09-07","arxiv_id":"1809.02256","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-source-domain-adaptation-with-mixture#ran","syntology_url":"https://syntology.ai/paper/1809.02256","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.02256"}},"official":{"repos":["jiangfeng1124/transfer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tafe-net-task-aware-feature-embeddings-for","slug":"tafe-net-task-aware-feature-embeddings-for","title":"Deep Mixture of Experts via Shallow Embedding","date":"2018-06-05","arxiv_id":"1806.01531","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tafe-net-task-aware-feature-embeddings-for#ran","syntology_url":"https://syntology.ai/paper/1806.01531","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.01531"}},"official":null}},{"url":"/paper/discontinuity-sensitive-optimal-control","slug":"discontinuity-sensitive-optimal-control","title":"Discontinuity-Sensitive Optimal Control Learning by Mixture of Experts","date":"2018-03-07","arxiv_id":"1803.02493","repositories_listed":1,"syntology":null},{"url":"/paper/granger-causal-attentive-mixtures-of-experts","slug":"granger-causal-attentive-mixtures-of-experts","title":"Granger-causal Attentive Mixtures of Experts: Learning Important Features with Neural Networks","date":"2018-02-06","arxiv_id":"1802.02195","repositories_listed":1,"syntology":null},{"url":"/paper/learning-gating-convnet-for-two-stream-based","slug":"learning-gating-convnet-for-two-stream-based","title":"Learning Gating ConvNet for Two-Stream based Methods in Action Recognition","date":"2017-09-12","arxiv_id":"1709.03655","repositories_listed":1,"syntology":null},{"url":"/paper/uts-submission-to-google-youtube-8m-challenge","slug":"uts-submission-to-google-youtube-8m-challenge","title":"UTS submission to Google YouTube-8M Challenge 2017","date":"2017-07-13","arxiv_id":"1707.04143","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-deep-recurrent-architecture-for","slug":"hierarchical-deep-recurrent-architecture-for","title":"Hierarchical Deep Recurrent Architecture for Video Understanding","date":"2017-07-11","arxiv_id":"1707.03296","repositories_listed":1,"syntology":null},{"url":"/paper/effective-approaches-to-batch-parallelization","slug":"effective-approaches-to-batch-parallelization","title":"Effective Approaches to Batch Parallelization for Dynamic Neural Network Architectures","date":"2017-07-08","arxiv_id":"1707.02402","repositories_listed":1,"syntology":null},{"url":"/paper/embarrassingly-parallel-inference-for","slug":"embarrassingly-parallel-inference-for","title":"Embarrassingly Parallel Inference for Gaussian Processes","date":"2017-02-27","arxiv_id":"1702.08420","repositories_listed":1,"syntology":null},{"url":"/paper/opponent-modeling-in-deep-reinforcement","slug":"opponent-modeling-in-deep-reinforcement","title":"Opponent Modeling in Deep Reinforcement Learning","date":"2016-09-18","arxiv_id":"1609.05559","repositories_listed":1,"syntology":null},{"url":null,"slug":"geminus-dual-aware-global-and-scene-adaptive","title":"GEMINUS: Dual-aware Global and Scene-Adaptive Mixture-of-Experts for End-to-End Autonomous Driving","date":"2025-07-19","arxiv_id":"2507.14456","repositories_listed":0,"syntology":null},{"url":null,"slug":"r-2moe-redundancy-removal-mixture-of-experts","title":"R^2MoE: Redundancy-Removal Mixture of Experts for Lifelong Concept Learning","date":"2025-07-17","arxiv_id":"2507.13107","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-of-experts-in-large-language-models","title":"Mixture of Experts in Large Language Models","date":"2025-07-15","arxiv_id":"2507.11181","repositories_listed":0,"syntology":null},{"url":null,"slug":"inter2former-dynamic-hybrid-attention-for","title":"Inter2Former: Dynamic Hybrid Attention for Efficient High-Precision Interactive","date":"2025-07-13","arxiv_id":"2507.09612","repositories_listed":0,"syntology":null},{"url":null,"slug":"kat-v1-kwai-autothink-technical-report","title":"KAT-V1: Kwai-AutoThink Technical Report","date":"2025-07-11","arxiv_id":"2507.08297","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-training-of-large-scale-ai-models","title":"Efficient Training of Large-Scale AI Models Through Federated Mixture-of-Experts: A System-Level Approach","date":"2025-07-08","arxiv_id":"2507.05685","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-quality-assessment-model-based-on","title":"Speech Quality Assessment Model Based on Mixture of Experts: System-Level Performance Enhancement and Utterance-Level Challenge Analysis","date":"2025-07-08","arxiv_id":"2507.06116","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-you-have-is-what-you-track-adaptive-and","title":"What You Have is What You Track: Adaptive and Robust Multimodal Tracking","date":"2025-07-08","arxiv_id":"2507.05899","repositories_listed":0,"syntology":null},{"url":null,"slug":"ugg-reid-uncertainty-guided-graph-model-for","title":"UGG-ReID: Uncertainty-Guided Graph Model for Multi-Modal Object Re-Identification","date":"2025-07-07","arxiv_id":"2507.04638","repositories_listed":0,"syntology":null},{"url":null,"slug":"sub-moe-efficient-mixture-of-expert-llms","title":"Sub-MoE: Efficient Mixture-of-Expert LLMs Compression via Subspace Expert Merging","date":"2025-06-29","arxiv_id":"2506.23266","repositories_listed":0,"syntology":null},{"url":null,"slug":"eva-mixture-of-experts-semantic-variant","title":"EVA: Mixture-of-Experts Semantic Variant Alignment for Compositional Zero-Shot Learning","date":"2025-06-26","arxiv_id":"2506.20986","repositories_listed":0,"syntology":null},{"url":null,"slug":"little-by-little-continual-learning-via-self","title":"Little By Little: Continual Learning via Self-Activated Sparse Mixture-of-Rank Adaptive Learning","date":"2025-06-26","arxiv_id":"2506.21035","repositories_listed":0,"syntology":null},{"url":null,"slug":"opportunistic-osteoporosis-diagnosis-via","title":"Opportunistic Osteoporosis Diagnosis via Texture-Preserving Self-Supervision, Mixture of Experts and Multi-Task Integration","date":"2025-06-25","arxiv_id":"2506.20282","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-audio-centric-multi-task-learning","title":"An Audio-centric Multi-task Learning Framework for Streaming Ads Targeting on Spotify","date":"2025-06-23","arxiv_id":"2506.18735","repositories_listed":0,"syntology":null},{"url":null,"slug":"security-assessment-of-deepseek-and-gpt","title":"Security Assessment of DeepSeek and GPT Series Models against Jailbreak Attacks","date":"2025-06-23","arxiv_id":"2506.18543","repositories_listed":0,"syntology":null},{"url":null,"slug":"safex-analyzing-vulnerabilities-of-moe-based","title":"SAFEx: Analyzing Vulnerabilities of MoE-Based LLMs via Stable Safety-critical Expert Identification","date":"2025-06-20","arxiv_id":"2506.17368","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-speaker-diarization-with-mixture-of","title":"Exploring Speaker Diarization with Mixture of Experts","date":"2025-06-17","arxiv_id":"2506.14750","repositories_listed":0,"syntology":null},{"url":null,"slug":"lora-mixer-coordinate-modular-lora-experts","title":"LoRA-Mixer: Coordinate Modular LoRA Experts Through Serial Attention Routing","date":"2025-06-17","arxiv_id":"2507.00029","repositories_listed":0,"syntology":null},{"url":null,"slug":"mote-mixture-of-ternary-experts-for-memory","title":"MoTE: Mixture of Ternary Experts for Memory-efficient Large Multimodal Models","date":"2025-06-17","arxiv_id":"2506.14435","repositories_listed":0,"syntology":null},{"url":null,"slug":"neuromoe-a-transformer-based-mixture-of","title":"NeuroMoE: A Transformer-Based Mixture-of-Experts Framework for Multi-Modal Neurological Disorder Classification","date":"2025-06-17","arxiv_id":"2506.14970","repositories_listed":0,"syntology":null},{"url":null,"slug":"ring-lite-scalable-reasoning-via-c3po","title":"Ring-lite: Scalable Reasoning via C3PO-Stabilized Reinforcement Learning for LLMs","date":"2025-06-17","arxiv_id":"2506.14731","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-intelligence-designing-data-centers","title":"Scaling Intelligence: Designing Data Centers for Next-Gen Language Models","date":"2025-06-17","arxiv_id":"2506.15006","repositories_listed":0,"syntology":null},{"url":null,"slug":"single-example-learning-in-a-mixture-of-gpdms","title":"Single-Example Learning in a Mixture of GPDMs with Latent Geometries","date":"2025-06-17","arxiv_id":"2506.14563","repositories_listed":0,"syntology":null},{"url":null,"slug":"utility-driven-speculative-decoding-for","title":"Utility-Driven Speculative Decoding for Mixture-of-Experts","date":"2025-06-17","arxiv_id":"2506.20675","repositories_listed":0,"syntology":null},{"url":null,"slug":"eaquant-enhancing-post-training-quantization","title":"EAQuant: Enhancing Post-Training Quantization for MoE Models via Expert-Aware Optimization","date":"2025-06-16","arxiv_id":"2506.13329","repositories_listed":0,"syntology":null},{"url":null,"slug":"load-balancing-mixture-of-experts-with","title":"Load Balancing Mixture of Experts with Similarity Preserving Routers","date":"2025-06-16","arxiv_id":"2506.14038","repositories_listed":0,"syntology":null},{"url":null,"slug":"serving-large-language-models-on-huawei","title":"Serving Large Language Models on Huawei CloudMatrix384","date":"2025-06-15","arxiv_id":"2506.12708","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimus-3-towards-generalist-multimodal","title":"Optimus-3: Towards Generalist Multimodal Minecraft Agents with Scalable Task Experts","date":"2025-06-12","arxiv_id":"2506.10357","repositories_listed":0,"syntology":null},{"url":null,"slug":"gigachat-family-efficient-russian-language","title":"GigaChat Family: Efficient Russian Language Modeling Through Mixture of Experts Architecture","date":"2025-06-11","arxiv_id":"2506.09440","repositories_listed":0,"syntology":null},{"url":null,"slug":"medmoe-modality-specialized-mixture-of","title":"MedMoE: Modality-Specialized Mixture of Experts for Medical Vision-Language Understanding","date":"2025-06-10","arxiv_id":"2506.08356","repositories_listed":0,"syntology":null},{"url":null,"slug":"m2restore-mixture-of-experts-based-mamba-cnn","title":"M2Restore: Mixture-of-Experts-based Mamba-CNN Fusion Framework for All-in-One Image Restoration","date":"2025-06-09","arxiv_id":"2506.07814","repositories_listed":0,"syntology":null},{"url":null,"slug":"mira-medical-time-series-foundation-model-for","title":"MIRA: Medical Time Series Foundation Model for Real-World Health Data","date":"2025-06-09","arxiv_id":"2506.07584","repositories_listed":0,"syntology":null},{"url":null,"slug":"moe-gps-guidlines-for-prediction-strategy-for","title":"MoE-GPS: Guidlines for Prediction Strategy for Dynamic Expert Duplication in MoE Load Balancing","date":"2025-06-09","arxiv_id":"2506.07366","repositories_listed":0,"syntology":null},{"url":null,"slug":"breaking-data-silos-towards-open-and-scalable","title":"Breaking Data Silos: Towards Open and Scalable Mobility Foundation Models via Generative Continual Learning","date":"2025-06-07","arxiv_id":"2506.06694","repositories_listed":0,"syntology":null},{"url":null,"slug":"smar-soft-modality-aware-routing-strategy-for","title":"SMAR: Soft Modality-Aware Routing Strategy for MoE-based Multimodal Large Language Models Preserving Language Capabilities","date":"2025-06-06","arxiv_id":"2506.06406","repositories_listed":0,"syntology":null},{"url":null,"slug":"lifelong-evolution-collaborative-learning","title":"Lifelong Evolution: Collaborative Learning between Large and Small Language Models for Continuous Emergent Fake News Detection","date":"2025-06-05","arxiv_id":"2506.04739","repositories_listed":0,"syntology":null},{"url":null,"slug":"brain-like-processing-pathways-form-in-models","title":"Brain-Like Processing Pathways Form in Models With Heterogeneous Experts","date":"2025-06-03","arxiv_id":"2506.02813","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-multimodal-continual-instruction","title":"Enhancing Multimodal Continual Instruction Tuning with BranchLoRA","date":"2025-05-31","arxiv_id":"2506.02041","repositories_listed":0,"syntology":null},{"url":null,"slug":"decoding-knowledge-attribution-in-mixture-of","title":"Decoding Knowledge Attribution in Mixture-of-Experts: A Framework of Basic-Refinement Collaboration and Efficiency Analysis","date":"2025-05-30","arxiv_id":"2505.24593","repositories_listed":0,"syntology":null},{"url":null,"slug":"gradpower-powering-gradients-for-faster","title":"GradPower: Powering Gradients for Faster Language Model Pre-Training","date":"2025-05-30","arxiv_id":"2505.24275","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-of-experts-for-personalized-and","title":"Mixture-of-Experts for Personalized and Semantic-Aware Next Location Prediction","date":"2025-05-30","arxiv_id":"2505.24597","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-expressive-power-of-mixture-of-experts","title":"On the Expressive Power of Mixture-of-Experts for Structured Complex Tasks","date":"2025-05-30","arxiv_id":"2505.24205","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-10016","title":"A Survey of Generative Categories and Techniques in Multimodal Large Language Models","date":"2025-05-29","arxiv_id":"2506.10016","repositories_listed":0,"syntology":null},{"url":null,"slug":"noise-robustness-through-noise-asymmetric","title":"Noise-Robustness Through Noise: Asymmetric LoRA Adaption with Poisoning Expert","date":"2025-05-29","arxiv_id":"2505.23868","repositories_listed":0,"syntology":null},{"url":null,"slug":"point-moe-towards-cross-domain-generalization","title":"Point-MoE: Towards Cross-Domain Generalization in 3D Semantic Segmentation via Mixture-of-Experts","date":"2025-05-29","arxiv_id":"2505.23926","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-uncertainty-estimation-and","title":"Revisiting Uncertainty Estimation and Calibration of Large Language Models","date":"2025-05-29","arxiv_id":"2505.23854","repositories_listed":0,"syntology":null},{"url":null,"slug":"two-is-better-than-one-rotations-scale-loras","title":"Two Is Better Than One: Rotations Scale LoRAs","date":"2025-05-29","arxiv_id":"2505.23184","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-expert-specialization-for-better","title":"Advancing Expert Specialization for Better MoE","date":"2025-05-28","arxiv_id":"2505.22323","repositories_listed":0,"syntology":null},{"url":null,"slug":"evomoe-expert-evolution-in-mixture-of-experts","title":"EvoMoE: Expert Evolution in Mixture of Experts for Multimodal Large Language Models","date":"2025-05-28","arxiv_id":"2505.23830","repositories_listed":0,"syntology":null},{"url":null,"slug":"forcevla-enhancing-vla-models-with-a-force","title":"ForceVLA: Enhancing VLA Models with a Force-aware MoE for Contact-rich Manipulation","date":"2025-05-28","arxiv_id":"2505.22159","repositories_listed":0,"syntology":null},{"url":null,"slug":"moe-gyro-self-supervised-over-range","title":"MoE-Gyro: Self-Supervised Over-Range Reconstruction and Denoising for MEMS Gyroscopes","date":"2025-05-27","arxiv_id":"2506.06318","repositories_listed":0,"syntology":null},{"url":null,"slug":"moesd-unveil-speculative-decoding-s-potential","title":"MoESD: Unveil Speculative Decoding's Potential for Accelerating Sparse MoE","date":"2025-05-26","arxiv_id":"2505.19645","repositories_listed":0,"syntology":null},{"url":null,"slug":"mosaic-data-free-knowledge-distillation-via","title":"Mosaic: Data-Free Knowledge Distillation via Mixture-of-Experts for Heterogeneous Distributed Environments","date":"2025-05-26","arxiv_id":"2505.19699","repositories_listed":0,"syntology":null},{"url":null,"slug":"next-multi-grained-mixture-of-experts-via","title":"NEXT: Multi-Grained Mixture of Experts via Text-Modulation for Multi-Modal Object Re-ID","date":"2025-05-26","arxiv_id":"2505.20001","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-dynamical-systems-learning-with","title":"Integrating Dynamical Systems Learning with Foundational Models: A Meta-Evolutionary AI Framework for Clinical Trials","date":"2025-05-25","arxiv_id":"2506.14782","repositories_listed":0,"syntology":null},{"url":null,"slug":"m-moe-test-time-pruning-as-micro-grained","title":"$μ$-MoE: Test-Time Pruning as Micro-Grained Mixture-of-Experts","date":"2025-05-24","arxiv_id":"2505.18451","repositories_listed":0,"syntology":null},{"url":null,"slug":"mod-adapter-tuning-free-and-versatile-multi","title":"Mod-Adapter: Tuning-Free and Versatile Multi-concept Personalization via Modulation Adapter","date":"2025-05-24","arxiv_id":"2505.18612","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-minimax-estimation-of-parameters-in","title":"On Minimax Estimation of Parameters in Softmax-Contaminated Mixture of Experts","date":"2025-05-24","arxiv_id":"2505.18455","repositories_listed":0,"syntology":null},{"url":null,"slug":"trajmoe-spatially-aware-mixture-of-experts","title":"TrajMoE: Spatially-Aware Mixture of Experts for Unified Human Mobility Modeling","date":"2025-05-24","arxiv_id":"2505.18670","repositories_listed":0,"syntology":null},{"url":null,"slug":"evidencemoe-a-physics-guided-mixture-of","title":"EvidenceMoE: A Physics-Guided Mixture-of-Experts with Evidential Critics for Advancing Fluorescence Light Detection and Ranging in Scattering Media","date":"2025-05-23","arxiv_id":"2505.21532","repositories_listed":0,"syntology":null},{"url":"/paper/drivemoe-mixture-of-experts-for-vision","slug":"drivemoe-mixture-of-experts-for-vision","title":"DriveMoE: Mixture-of-Experts for Vision-Language-Action Model in End-to-End Autonomous Driving","date":"2025-05-22","arxiv_id":"2505.16278","repositories_listed":0,"syntology":null},{"url":null,"slug":"dualcomp-end-to-end-learning-of-a-unified","title":"DualComp: End-to-End Learning of a Unified Dual-Modality Lossless Compressor","date":"2025-05-22","arxiv_id":"2505.16256","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-data-driven-mixture-of-expert","title":"Efficient Data Driven Mixture-of-Expert Extraction from Trained Networks","date":"2025-05-21","arxiv_id":"2505.15414","repositories_listed":0,"syntology":null},{"url":null,"slug":"hunyuan-turbos-advancing-large-language","title":"Hunyuan-TurboS: Advancing Large Language Models through Mamba-Transformer Synergy and Adaptive Chain-of-Thought","date":"2025-05-21","arxiv_id":"2505.15431","repositories_listed":0,"syntology":null},{"url":"/paper/more-brain-routed-mixture-of-experts-for","slug":"more-brain-routed-mixture-of-experts-for","title":"MoRE-Brain: Routed Mixture of Experts for Interpretable and Generalizable Cross-Subject fMRI Visual Decoding","date":"2025-05-21","arxiv_id":"2505.15946","repositories_listed":0,"syntology":{"n":4,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/more-brain-routed-mixture-of-experts-for#ran","syntology_url":"https://syntology.ai/paper/2505.15946","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15946"}},"official":null}},{"url":null,"slug":"time-tracker-mixture-of-experts-enhanced","title":"Time Tracker: Mixture-of-Experts-Enhanced Foundation Time Series Forecasting Model with Decoupled Training Pipelines","date":"2025-05-21","arxiv_id":"2505.15151","repositories_listed":0,"syntology":null},{"url":null,"slug":"balanced-and-elastic-end-to-end-training-of","title":"Balanced and Elastic End-to-end Training of Dynamic LLMs","date":"2025-05-20","arxiv_id":"2505.14864","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficientllm-efficiency-in-large-language","title":"EfficientLLM: Efficiency in Large Language Models","date":"2025-05-20","arxiv_id":"2505.13840","repositories_listed":0,"syntology":null},{"url":null,"slug":"fuximt-sparsifying-large-language-models-for","title":"FuxiMT: Sparsifying Large Language Models for Chinese-Centric Multilingual Machine Translation","date":"2025-05-20","arxiv_id":"2505.14256","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-mixture-of-low-rank-experts-for","title":"Multimodal Mixture of Low-Rank Experts for Sentiment Analysis and Emotion Recognition","date":"2025-05-20","arxiv_id":"2505.14143","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-and-enhancing-llm-based-avsr-a-sparse","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","date":"2025-05-20","arxiv_id":"2505.14336","repositories_listed":0,"syntology":null},{"url":null,"slug":"stpr-spatiotemporal-preservation-and-routing","title":"StPR: Spatiotemporal Preservation and Routing for Exemplar-Free Video Class-Incremental Learning","date":"2025-05-20","arxiv_id":"2505.13997","repositories_listed":0,"syntology":null},{"url":null,"slug":"thor-moe-hierarchical-task-guided-and-context","title":"THOR-MoE: Hierarchical Task-Guided and Context-Responsive Routing for Neural Machine Translation","date":"2025-05-20","arxiv_id":"2505.14173","repositories_listed":0,"syntology":null},{"url":null,"slug":"two-experts-are-all-you-need-for-steering","title":"Two Experts Are All You Need for Steering Thinking: Reinforcing Cognitive Effort in MoE Reasoning Models Without Additional Training","date":"2025-05-20","arxiv_id":"2505.14681","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-selection-for-gaussian-gated-gaussian","title":"Model Selection for Gaussian-gated Gaussian Mixture of Experts Using Dendrograms of Mixing Measures","date":"2025-05-19","arxiv_id":"2505.13052","repositories_listed":0,"syntology":null},{"url":null,"slug":"seeing-the-unseen-how-emoe-unveils-bias-in","title":"Seeing the Unseen: How EMoE Unveils Bias in Text-to-Image Diffusion Models","date":"2025-05-19","arxiv_id":"2505.13273","repositories_listed":0,"syntology":null},{"url":null,"slug":"true-zero-shot-inference-of-dynamical-systems","title":"True Zero-Shot Inference of Dynamical Systems Preserving Long-Term Statistics","date":"2025-05-19","arxiv_id":"2505.13192","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-coverage-in-combined-prediction","title":"Improving Coverage in Combined Prediction Sets with Weighted p-values","date":"2025-05-17","arxiv_id":"2505.11785","repositories_listed":0,"syntology":null},{"url":null,"slug":"mingle-mixtures-of-null-space-gated-low-rank","title":"MINGLE: Mixtures of Null-Space Gated Low-Rank Experts for Test-Time Continual Model Merging","date":"2025-05-17","arxiv_id":"2505.11883","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-merging-in-pre-training-of-large","title":"Model Merging in Pre-training of Large Language Models","date":"2025-05-17","arxiv_id":"2505.12082","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-10860","title":"On DeepSeekMoE: Statistical Benefits of Shared Experts and Normalized Sigmoid Gating","date":"2025-05-16","arxiv_id":"2505.10860","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-11085","title":"A Fast Kernel-based Conditional Independence test with Application to Causal Discovery","date":"2025-05-16","arxiv_id":"2505.11085","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-11415","title":"MoE-CAP: Benchmarking Cost, Accuracy and Performance of Sparse Mixture-of-Experts Systems","date":"2025-05-16","arxiv_id":"2505.11415","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-11432","title":"MegaScale-MoE: Large-Scale Communication-Efficient Training of Mixture-of-Experts Models in Production","date":"2025-05-16","arxiv_id":"2505.11432","repositories_listed":0,"syntology":null}],"record_sha256":"a1826d36a52c96d31b908001605ba1e658e312582ce1fa56b87b7189256f0b46","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}