{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/mixture-of-experts/papers/7","list_of":"/task/mixture-of-experts","task":"Mixture-of-Experts","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":7,"pages_in_order":14,"rows_per_page":100,"rows":[601,700],"of":1312,"counts":{"archive_papers_tagged":1312,"with_a_code_link":516,"where_syntology_ran_a_sample":216,"not_listed_spam_title":0,"listed":1312,"listed_where_code_ran":216,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":184,"every_run_a_failure_of_syntologys_instrument":32,"listed_with_a_run_with_no_instrument_failure":184,"listed_every_run_a_failure_of_syntologys_instrument":32,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/mixture-of-experts","prev":"/task/mixture-of-experts/papers/6","next":"/task/mixture-of-experts/papers/8","papers":[{"url":null,"slug":"insights-into-deepseek-v3-scaling-challenges","title":"Insights into DeepSeek-V3: Scaling Challenges and Reflections on Hardware for AI Architectures","date":"2025-05-14","arxiv_id":"2505.09343","repositories_listed":0,"syntology":null},{"url":null,"slug":"am-thinking-v1-advancing-the-frontier-of","title":"AM-Thinking-v1: Advancing the Frontier of Reasoning at 32B Scale","date":"2025-05-13","arxiv_id":"2505.08311","repositories_listed":0,"syntology":null},{"url":null,"slug":"pwc-moe-privacy-aware-wireless-collaborative","title":"PWC-MoE: Privacy-Aware Wireless Collaborative Mixture of Experts","date":"2025-05-13","arxiv_id":"2505.08719","repositories_listed":0,"syntology":null},{"url":null,"slug":"umoe-unifying-attention-and-ffn-with-shared","title":"UMoE: Unifying Attention and FFN with Shared Experts","date":"2025-05-12","arxiv_id":"2505.07260","repositories_listed":0,"syntology":null},{"url":null,"slug":"freqmoe-dynamic-frequency-enhancement-for","title":"FreqMoE: Dynamic Frequency Enhancement for Neural PDE Solvers","date":"2025-05-11","arxiv_id":"2505.06858","repositories_listed":0,"syntology":null},{"url":"/paper/seed1-5-vl-technical-report","slug":"seed1-5-vl-technical-report","title":"Seed1.5-VL Technical Report","date":"2025-05-11","arxiv_id":"2505.07062","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-power-of-fine-grained-experts-granularity","title":"The power of fine-grained experts: Granularity boosts expressivity in Mixture of Experts","date":"2025-05-11","arxiv_id":"2505.06839","repositories_listed":0,"syntology":null},{"url":null,"slug":"qos-efficient-serving-of-multiple-mixture-of","title":"QoS-Efficient Serving of Multiple Mixture-of-Expert LLMs Using Partial Runtime Reconfiguration","date":"2025-05-10","arxiv_id":"2505.06481","repositories_listed":0,"syntology":null},{"url":null,"slug":"floe-on-the-fly-moe-inference","title":"FloE: On-the-Fly MoE Inference on Memory-constrained GPU","date":"2025-05-09","arxiv_id":"2505.05950","repositories_listed":0,"syntology":null},{"url":null,"slug":"divide-and-conquer-cold-start-bundle","title":"Divide-and-Conquer: Cold-Start Bundle Recommendation via Mixture of Diffusion Experts","date":"2025-05-08","arxiv_id":"2505.05035","repositories_listed":0,"syntology":null},{"url":null,"slug":"pangu-ultra-moe-how-to-train-your-big-moe-on","title":"Pangu Ultra MoE: How to Train Your Big MoE on Ascend NPUs","date":"2025-05-07","arxiv_id":"2505.04519","repositories_listed":0,"syntology":null},{"url":null,"slug":"stola-self-adaptive-touch-language-framework","title":"SToLa: Self-Adaptive Touch-Language Framework with Tactile Commonsense Reasoning in Open-Ended Scenarios","date":"2025-05-07","arxiv_id":"2505.04201","repositories_listed":0,"syntology":null},{"url":null,"slug":"3d-gaussian-splatting-data-compression-with","title":"3D Gaussian Splatting Data Compression with Mixture of Priors","date":"2025-05-06","arxiv_id":"2505.03310","repositories_listed":0,"syntology":null},{"url":null,"slug":"faster-moe-llm-inference-for-extremely-large","title":"Faster MoE LLM Inference for Extremely Large Models","date":"2025-05-06","arxiv_id":"2505.03531","repositories_listed":0,"syntology":null},{"url":null,"slug":"star-rec-making-peace-with-length-variance","title":"STAR-Rec: Making Peace with Length Variance and Pattern Diversity in Sequential Recommendation","date":"2025-05-06","arxiv_id":"2505.03484","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-smart-point-and-shoot-photography","title":"Towards Smart Point-and-Shoot Photography","date":"2025-05-06","arxiv_id":"2505.03638","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-deep-learning-empowered-beam","title":"Multimodal Deep Learning-Empowered Beam Prediction in Future THz ISAC Systems","date":"2025-05-05","arxiv_id":"2505.02381","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-llms-for-resource-constrained","title":"Optimizing LLMs for Resource-Constrained Environments: A Survey of Model Compression Techniques","date":"2025-05-05","arxiv_id":"2505.02309","repositories_listed":0,"syntology":null},{"url":null,"slug":"cocoafuse-beyond-mixtures-of-experts-via","title":"CoCoAFusE: Beyond Mixtures of Experts via Model Fusion","date":"2025-05-02","arxiv_id":"2505.01105","repositories_listed":0,"syntology":null},{"url":null,"slug":"perception-informed-neural-networks-beyond","title":"Perception-Informed Neural Networks: Beyond Physics-Informed Neural Networks","date":"2025-05-02","arxiv_id":"2505.03806","repositories_listed":0,"syntology":null},{"url":null,"slug":"cicada-cross-domain-interpretable-coding-for","title":"CICADA: Cross-Domain Interpretable Coding for Anomaly Detection and Adaptation in Multivariate Time Series","date":"2025-05-01","arxiv_id":"2505.00415","repositories_listed":0,"syntology":null},{"url":null,"slug":"moxe-mixture-of-xlstm-experts-with-entropy","title":"MoxE: Mixture of xLSTM Experts with Entropy-Aware Routing for Efficient Language Modeling","date":"2025-05-01","arxiv_id":"2505.01459","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-mixture-of-experts-training-with","title":"Accelerating Mixture-of-Experts Training with Adaptive Expert Replication","date":"2025-04-28","arxiv_id":"2504.19925","repositories_listed":0,"syntology":null},{"url":null,"slug":"pico-secure-transformers-via-robust-prompt","title":"PICO: Secure Transformers via Robust Prompt Isolation and Cybersecurity Oversight","date":"2025-04-26","arxiv_id":"2504.21029","repositories_listed":0,"syntology":null},{"url":"/paper/noesis-differentially-private-knowledge","slug":"noesis-differentially-private-knowledge","title":"NoEsis: Differentially Private Knowledge Transfer in Modular LLM Adaptation","date":"2025-04-25","arxiv_id":"2504.18147","repositories_listed":0,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/noesis-differentially-private-knowledge#ran","syntology_url":"https://syntology.ai/paper/2504.18147","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.18147"}},"official":null}},{"url":null,"slug":"badmoe-backdooring-mixture-of-experts-llms","title":"BadMoE: Backdooring Mixture-of-Experts LLMs via Optimizing Routing Triggers and Infecting Dormant Experts","date":"2025-04-24","arxiv_id":"2504.18598","repositories_listed":0,"syntology":null},{"url":null,"slug":"moe-parallel-folding-heterogeneous","title":"MoE Parallel Folding: Heterogeneous Parallelism Mappings for Efficient Large-Scale MoE Model Training with Megatron Core","date":"2025-04-21","arxiv_id":"2504.14960","repositories_listed":0,"syntology":null},{"url":null,"slug":"haeccity-open-vocabulary-scene-understanding","title":"HAECcity: Open-Vocabulary Scene Understanding of City-Scale Point Clouds with Superpoint Graph Clustering","date":"2025-04-18","arxiv_id":"2504.13590","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-type-context-aware-conversational","title":"Multi-Type Context-Aware Conversational Recommender Systems via Mixture-of-Experts","date":"2025-04-18","arxiv_id":"2504.13655","repositories_listed":0,"syntology":null},{"url":null,"slug":"d-2-moe-dual-routing-and-dynamic-scheduling","title":"D$^{2}$MoE: Dual Routing and Dynamic Scheduling for Efficient On-Device MoE-based LLM Serving","date":"2025-04-17","arxiv_id":"2504.15299","repositories_listed":0,"syntology":null},{"url":null,"slug":"trend-filtered-mixture-of-experts-for","title":"Trend Filtered Mixture of Experts for Automated Gating of High-Frequency Flow Cytometry Data","date":"2025-04-16","arxiv_id":"2504.12287","repositories_listed":0,"syntology":null},{"url":null,"slug":"unveiling-hidden-collaboration-within-mixture","title":"Unveiling Hidden Collaboration within Mixture-of-Experts in Large Language Models","date":"2025-04-16","arxiv_id":"2504.12359","repositories_listed":0,"syntology":null},{"url":"/paper/plasticity-aware-mixture-of-experts-for","slug":"plasticity-aware-mixture-of-experts-for","title":"Plasticity-Aware Mixture of Experts for Learning Under QoE Shifts in Adaptive Video Streaming","date":"2025-04-14","arxiv_id":"2504.09906","repositories_listed":0,"syntology":{"n":4,"n_ran":4,"n_constructed":3,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/plasticity-aware-mixture-of-experts-for#ran","syntology_url":"https://syntology.ai/paper/2504.09906","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.09906"}},"official":null}},{"url":null,"slug":"mixture-of-shape-experts-mose-end-to-end","title":"Mixture-of-Shape-Experts (MoSE): End-to-End Shape Dictionary Framework to Prompt SAM for Generalizable Medical Segmentation","date":"2025-04-13","arxiv_id":"2504.09601","repositories_listed":0,"syntology":null},{"url":null,"slug":"moe-lens-towards-the-hardware-limit-of-high","title":"MoE-Lens: Towards the Hardware Limit of High-Throughput MoE LLM Serving Under Resource Constraints","date":"2025-04-12","arxiv_id":"2504.09345","repositories_listed":0,"syntology":null},{"url":null,"slug":"regularized-infill-criteria-for-multi","title":"Regularized infill criteria for multi-objective Bayesian optimization with application to aircraft design","date":"2025-04-11","arxiv_id":"2504.08671","repositories_listed":0,"syntology":null},{"url":null,"slug":"routerkt-mixture-of-experts-for-knowledge","title":"RouterKT: Mixture-of-Experts for Knowledge Tracing","date":"2025-04-11","arxiv_id":"2504.08989","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-detection-of-fast-moving-celestial","title":"Adaptive Detection of Fast Moving Celestial Objects Using a Mixture of Experts and Physical-Inspired Neural Network","date":"2025-04-10","arxiv_id":"2504.07777","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-laws-for-native-multimodal-models","title":"Scaling Laws for Native Multimodal Models Scaling Laws for Native Multimodal Models","date":"2025-04-10","arxiv_id":"2504.07951","repositories_listed":0,"syntology":null},{"url":null,"slug":"seed1-5-thinking-advancing-superb-reasoning","title":"Seed1.5-Thinking: Advancing Superb Reasoning Models with Reinforcement Learning","date":"2025-04-10","arxiv_id":"2504.13914","repositories_listed":0,"syntology":null},{"url":null,"slug":"fedmerge-federated-personalization-via-model","title":"FedMerge: Federated Personalization via Model Merging","date":"2025-04-09","arxiv_id":"2504.06768","repositories_listed":0,"syntology":null},{"url":null,"slug":"holistic-capability-preservation-towards","title":"Holistic Capability Preservation: Towards Compact Yet Comprehensive Reasoning Models","date":"2025-04-09","arxiv_id":"2504.07158","repositories_listed":0,"syntology":null},{"url":null,"slug":"finding-fantastic-experts-in-moes-a-unified","title":"Finding Fantastic Experts in MoEs: A Unified Study for Expert Dropping Strategies and Observations","date":"2025-04-08","arxiv_id":"2504.05586","repositories_listed":0,"syntology":null},{"url":null,"slug":"hetermoe-efficient-training-of-mixture-of","title":"HeterMoE: Efficient Training of Mixture-of-Experts Models on Heterogeneous GPUs","date":"2025-04-04","arxiv_id":"2504.03871","repositories_listed":0,"syntology":null},{"url":null,"slug":"ringmoe-mixture-of-modality-experts-multi","title":"RingMoE: Mixture-of-Modality-Experts Multi-Modal Foundation Models for Universal Remote Sensing Image Interpretation","date":"2025-04-04","arxiv_id":"2504.03166","repositories_listed":0,"syntology":null},{"url":null,"slug":"megascale-infer-serving-mixture-of-experts-at","title":"MegaScale-Infer: Serving Mixture-of-Experts at Scale with Disaggregated Expert Parallelism","date":"2025-04-03","arxiv_id":"2504.02263","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-moe-efficiency-a-collaboration","title":"Advancing MoE Efficiency: A Collaboration-Constrained Routing (C2R) Strategy for Better Expert Parallelism Design","date":"2025-04-02","arxiv_id":"2504.01337","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-virtual-mixture-of-experts","title":"A Unified Virtual Mixture-of-Experts Framework:Enhanced Inference and Hallucination Mitigation in Single-Model System","date":"2025-04-01","arxiv_id":"2504.03739","repositories_listed":0,"syntology":null},{"url":null,"slug":"detecting-financial-fraud-with-hybrid-deep","title":"Detecting Financial Fraud with Hybrid Deep Learning: A Mix-of-Experts Approach to Sequential and Anomalous Patterns","date":"2025-04-01","arxiv_id":"2504.03750","repositories_listed":0,"syntology":null},{"url":null,"slug":"unimodal-driven-distillation-in-multimodal","title":"Unimodal-driven Distillation in Multimodal Emotion Recognition with Dynamic Fusion","date":"2025-03-31","arxiv_id":"2503.23721","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-of-routers","title":"Mixture of Routers","date":"2025-03-30","arxiv_id":"2503.23362","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-standard-moe-mixture-of-latent-experts","title":"Beyond Standard MoE: Mixture of Latent Experts for Resource-Efficient Language Models","date":"2025-03-29","arxiv_id":"2503.23100","repositories_listed":0,"syntology":null},{"url":null,"slug":"s2moe-robust-sparse-mixture-of-experts-via","title":"S2MoE: Robust Sparse Mixture of Experts via Stochastic Learning","date":"2025-03-29","arxiv_id":"2503.23007","repositories_listed":0,"syntology":null},{"url":null,"slug":"sparse-mixture-of-experts-as-unified","title":"Sparse Mixture of Experts as Unified Competitive Learning","date":"2025-03-29","arxiv_id":"2503.22996","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-mixture-of-experts-redundancy","title":"Exploiting Mixture-of-Experts Redundancy Unlocks Multimodal Generative Abilities","date":"2025-03-28","arxiv_id":"2503.22517","repositories_listed":0,"syntology":null},{"url":null,"slug":"imedimage-technical-report","title":"iMedImage Technical Report","date":"2025-03-27","arxiv_id":"2503.21836","repositories_listed":0,"syntology":null},{"url":null,"slug":"llava-cmoe-towards-continual-mixture-of","title":"LLaVA-CMoE: Towards Continual Mixture of Experts for Large Vision-Language Models","date":"2025-03-27","arxiv_id":"2503.21227","repositories_listed":0,"syntology":null},{"url":null,"slug":"rocketppa-ultra-fast-llm-based-ppa-estimator","title":"RocketPPA: Code-Level Power, Performance, and Area Prediction via LLM and Mixture of Experts","date":"2025-03-27","arxiv_id":"2503.21971","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-multi-modal-models-with","title":"Enhancing Multi-modal Models with Heterogeneous MoE Adapters for Fine-tuning","date":"2025-03-26","arxiv_id":"2503.20633","repositories_listed":0,"syntology":null},{"url":null,"slug":"mole-vla-dynamic-layer-skipping-vision","title":"MoLe-VLA: Dynamic Layer-skipping Vision Language Action Model via Mixture-of-Layers for Efficient Robot Manipulation","date":"2025-03-26","arxiv_id":"2503.20384","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-scaling-laws-for-efficiency-gains-in","title":"Optimal Scaling Laws for Efficiency Gains in a Theoretical Transformer-Augmented Sectional MoE Framework","date":"2025-03-26","arxiv_id":"2503.20750","repositories_listed":0,"syntology":null},{"url":null,"slug":"reasoning-beyond-limits-advances-and-open","title":"Reasoning Beyond Limits: Advances and Open Problems for LLMs","date":"2025-03-26","arxiv_id":"2503.22732","repositories_listed":0,"syntology":null},{"url":null,"slug":"biprompt-sam-enhancing-image-segmentation-via","title":"BiPrompt-SAM: Enhancing Image Segmentation via Explicit Selection between Point and Text Prompts","date":"2025-03-25","arxiv_id":"2503.19769","repositories_listed":0,"syntology":null},{"url":null,"slug":"m-2-cd-a-unified-multimodal-framework-for","title":"M$^2$CD: A Unified MultiModal Framework for Optical-SAR Change Detection with Mixture of Experts and Self-Distillation","date":"2025-03-25","arxiv_id":"2503.19406","repositories_listed":0,"syntology":null},{"url":null,"slug":"resilient-sensor-fusion-under-adverse-sensor","title":"Resilient Sensor Fusion under Adverse Sensor Failures via Multi-Modal Expert Fusion","date":"2025-03-25","arxiv_id":"2503.19776","repositories_listed":0,"syntology":null},{"url":null,"slug":"galaxy-walker-geometry-aware-vlms-for-galaxy","title":"Galaxy Walker: Geometry-aware VLMs For Galaxy-scale Understanding","date":"2025-03-24","arxiv_id":"2503.18578","repositories_listed":0,"syntology":null},{"url":null,"slug":"expertrag-efficient-rag-with-mixture-of","title":"ExpertRAG: Efficient RAG with Mixture of Experts -- Optimizing Context Retrieval for Adaptive LLM Responses","date":"2025-03-23","arxiv_id":"2504.08744","repositories_listed":0,"syntology":null},{"url":null,"slug":"every-sample-matters-leveraging-mixture-of","title":"Every Sample Matters: Leveraging Mixture-of-Experts and High-Quality Data for Efficient and Accurate Code LLM","date":"2025-03-22","arxiv_id":"2503.17793","repositories_listed":0,"syntology":null},{"url":null,"slug":"expert-race-a-flexible-routing-strategy-for","title":"Expert Race: A Flexible Routing Strategy for Scaling Diffusion Transformer with Mixture of Experts","date":"2025-03-20","arxiv_id":"2503.16057","repositories_listed":0,"syntology":null},{"url":null,"slug":"unicorn-latent-diffusion-based-unified","title":"UniCoRN: Latent Diffusion-based Unified Controllable Image Restoration Network across Multiple Degradations","date":"2025-03-20","arxiv_id":"2503.15868","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-moe-based-large-language-model-for","title":"Leveraging MoE-based Large Language Model for Zero-Shot Multi-Task Semantic Communication","date":"2025-03-19","arxiv_id":"2503.15722","repositories_listed":0,"syntology":null},{"url":null,"slug":"semeval-2025-task-1-admire-advancing","title":"SemEval-2025 Task 1: AdMIRe -- Advancing Multimodal Idiomaticity Representation","date":"2025-03-19","arxiv_id":"2503.15358","repositories_listed":0,"syntology":null},{"url":null,"slug":"core-periphery-principle-guided-state-space","title":"Core-Periphery Principle Guided State Space Model for Functional Connectome Classification","date":"2025-03-18","arxiv_id":"2503.14655","repositories_listed":0,"syntology":null},{"url":null,"slug":"mast-pro-dynamic-mixture-of-experts-for","title":"MAST-Pro: Dynamic Mixture-of-Experts for Adaptive Segmentation of Pan-Tumors with Knowledge-Driven Prompts","date":"2025-03-18","arxiv_id":"2503.14355","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-mixture-of-experts-learning-for-1","title":"Adaptive Mixture of Low-Rank Experts for Robust Audio Spoofing Detection","date":"2025-03-15","arxiv_id":"2503.12010","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-deepseek-models-key-innovative","title":"A Review of DeepSeek Models' Key Innovative Techniques","date":"2025-03-14","arxiv_id":"2503.11486","repositories_listed":0,"syntology":null},{"url":null,"slug":"dflmoe-decentralized-federated-learning-via","title":"dFLMoE: Decentralized Federated Learning via Mixture of Experts for Medical Data Analysis","date":"2025-03-13","arxiv_id":"2503.10412","repositories_listed":0,"syntology":null},{"url":null,"slug":"ensemble-learning-for-large-language-models","title":"Ensemble Learning for Large Language Models in Text and Code Generation: A Survey","date":"2025-03-13","arxiv_id":"2503.13505","repositories_listed":0,"syntology":null},{"url":null,"slug":"astrea-a-moe-based-visual-understanding-model","title":"Astrea: A MOE-based Visual Understanding Model with Progressive Alignment","date":"2025-03-12","arxiv_id":"2503.09445","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-operator-level-parallelism-planning","title":"Automatic Operator-level Parallelism Planning for Distributed Deep Learning -- A Mixed-Integer Programming Approach","date":"2025-03-12","arxiv_id":"2503.09357","repositories_listed":0,"syntology":null},{"url":null,"slug":"double-stage-feature-level-clustering-based","title":"Double-Stage Feature-Level Clustering-Based Mixture of Experts Framework","date":"2025-03-12","arxiv_id":"2503.09504","repositories_listed":0,"syntology":null},{"url":null,"slug":"favchat-unlocking-fine-grained-facail-video","title":"FaVChat: Unlocking Fine-Grained Facail Video Understanding with Multimodal Large Language Models","date":"2025-03-12","arxiv_id":"2503.09158","repositories_listed":0,"syntology":null},{"url":null,"slug":"priority-aware-preemptive-scheduling-for","title":"Priority-Aware Preemptive Scheduling for Mixed-Priority Workloads in MoE Inference","date":"2025-03-12","arxiv_id":"2503.09304","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-moe-model-inference-with-expert","title":"Accelerating MoE Model Inference with Expert Sharding","date":"2025-03-11","arxiv_id":"2503.08467","repositories_listed":0,"syntology":null},{"url":null,"slug":"moe-loco-mixture-of-experts-for-multitask","title":"MoE-Loco: Mixture of Experts for Multitask Locomotion","date":"2025-03-11","arxiv_id":"2503.08564","repositories_listed":0,"syntology":null},{"url":null,"slug":"more-unlocking-scalability-in-reinforcement","title":"MoRE: Unlocking Scalability in Reinforcement Learning for Quadruped Vision-Language-Action Models","date":"2025-03-11","arxiv_id":"2503.08007","repositories_listed":0,"syntology":null},{"url":null,"slug":"uni-textbf-f-2-ace-fine-grained-face","title":"Uni$\\textbf{F}^2$ace: Fine-grained Face Understanding and Generation with Unified Multimodal Models","date":"2025-03-11","arxiv_id":"2503.08120","repositories_listed":0,"syntology":null},{"url":null,"slug":"emoe-task-aware-memory-efficient-mixture-of","title":"eMoE: Task-aware Memory Efficient Mixture-of-Experts-Based (MoE) Model Inference","date":"2025-03-10","arxiv_id":"2503.06823","repositories_listed":0,"syntology":null},{"url":null,"slug":"gm-moe-low-light-enhancement-with-gated","title":"GM-MoE: Low-Light Enhancement with Gated-Mechanism Mixture-of-Experts","date":"2025-03-10","arxiv_id":"2503.07417","repositories_listed":0,"syntology":null},{"url":null,"slug":"mofe-mixture-of-frozen-experts-architecture","title":"MoFE: Mixture of Frozen Experts Architecture","date":"2025-03-09","arxiv_id":"2503.06491","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-trustworthy-video-summarization","title":"A Novel Trustworthy Video Summarization Algorithm Through a Mixture of LoRA Experts","date":"2025-03-08","arxiv_id":"2503.06064","repositories_listed":0,"syntology":null},{"url":null,"slug":"mandarin-mixture-of-experts-framework-for","title":"MANDARIN: Mixture-of-Experts Framework for Dynamic Delirium and Coma Prediction in ICU Patients: Development and Validation of an Acute Brain Dysfunction Prediction Model","date":"2025-03-08","arxiv_id":"2503.06059","repositories_listed":0,"syntology":null},{"url":null,"slug":"moemoe-question-guided-dense-and-scalable","title":"MoEMoE: Question Guided Dense and Scalable Sparse Mixture-of-Expert for Multi-source Multi-modal Answering","date":"2025-03-08","arxiv_id":"2503.06296","repositories_listed":0,"syntology":null},{"url":null,"slug":"capacity-aware-inference-mitigating-the","title":"Capacity-Aware Inference: Mitigating the Straggler Effect in Mixture of Experts","date":"2025-03-07","arxiv_id":"2503.05066","repositories_listed":0,"syntology":null},{"url":null,"slug":"every-flop-counts-scaling-a-300b-mixture-of","title":"Every FLOP Counts: Scaling a 300B Mixture-of-Experts LING LLM without Premium GPUs","date":"2025-03-07","arxiv_id":"2503.05139","repositories_listed":0,"syntology":null},{"url":null,"slug":"fmt-a-multimodal-pneumonia-detection-model","title":"FMT:A Multimodal Pneumonia Detection Model Based on Stacking MOE Framework","date":"2025-03-07","arxiv_id":"2503.05626","repositories_listed":0,"syntology":null},{"url":null,"slug":"symbolic-mixture-of-experts-adaptive-skill","title":"Symbolic Mixture-of-Experts: Adaptive Skill-based Routing for Heterogeneous Reasoning","date":"2025-03-07","arxiv_id":"2503.05641","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-generalist-cross-domain-molecular-learning","title":"A Generalist Cross-Domain Molecular Learning Framework for Structure-Based Drug Discovery","date":"2025-03-06","arxiv_id":"2503.04362","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-pre-training-of-moes-how-robust-is","title":"Continual Pre-training of MoEs: How robust is your router?","date":"2025-03-06","arxiv_id":"2503.05029","repositories_listed":0,"syntology":null},{"url":null,"slug":"predictable-scale-part-i-optimal","title":"Predictable Scale: Part I -- Optimal Hyperparameter Scaling Law in Large Language Model Pretraining","date":"2025-03-06","arxiv_id":"2503.04715","repositories_listed":0,"syntology":null}],"record_sha256":"370e02fa844b0a9c934eb00cb49762048c80b64424d0748da96d5f6f810b80d4","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}