{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/mixture-of-experts/papers/9","list_of":"/task/mixture-of-experts","task":"Mixture-of-Experts","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":9,"pages_in_order":14,"rows_per_page":100,"rows":[801,900],"of":1312,"counts":{"archive_papers_tagged":1312,"with_a_code_link":516,"where_syntology_ran_a_sample":216,"not_listed_spam_title":0,"listed":1312,"listed_where_code_ran":216,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":184,"every_run_a_failure_of_syntologys_instrument":32,"listed_with_a_run_with_no_instrument_failure":184,"listed_every_run_a_failure_of_syntologys_instrument":32,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/mixture-of-experts","prev":"/task/mixture-of-experts/papers/8","next":"/task/mixture-of-experts/papers/10","papers":[{"url":null,"slug":"yi-lightning-technical-report","title":"Yi-Lightning Technical Report","date":"2024-12-02","arxiv_id":"2412.01253","repositories_listed":0,"syntology":null},{"url":null,"slug":"himoe-heterogeneity-informed-mixture-of","title":"HiMoE: Heterogeneity-Informed Mixture-of-Experts for Fair Spatial-Temporal Forecasting","date":"2024-11-30","arxiv_id":"2412.00316","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-of-experts-for-node-classification","title":"Mixture of Experts for Node Classification","date":"2024-11-30","arxiv_id":"2412.00418","repositories_listed":0,"syntology":null},{"url":null,"slug":"mqfl-fhe-multimodal-quantum-federated","title":"MQFL-FHE: Multimodal Quantum Federated Learning Framework with Fully Homomorphic Encryption","date":"2024-11-30","arxiv_id":"2412.01858","repositories_listed":0,"syntology":null},{"url":null,"slug":"lavide-a-language-vision-discriminator-for","title":"LaVIDE: A Language-Vision Discriminator for Detecting Changes in Satellite Image with Map References","date":"2024-11-29","arxiv_id":"2411.19758","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-effectiveness-of-discrete","title":"On the effectiveness of discrete representations in sparse mixture of experts","date":"2024-11-28","arxiv_id":"2411.19402","repositories_listed":0,"syntology":null},{"url":"/paper/complexity-experts-are-task-discriminative","slug":"complexity-experts-are-task-discriminative","title":"Complexity Experts are Task-Discriminative Learners for Any Image Restoration","date":"2024-11-27","arxiv_id":"2411.18466","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-of-cache-conditional-experts-for","title":"Mixture of Cache-Conditional Experts for Efficient Mobile Device Inference","date":"2024-11-27","arxiv_id":"2412.00099","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-of-experts-in-image-classification","title":"Mixture of Experts in Image Classification: What's the Sweet Spot?","date":"2024-11-27","arxiv_id":"2411.18322","repositories_listed":0,"syntology":null},{"url":null,"slug":"uoe-unlearning-one-expert-is-enough-for","title":"UOE: Unlearning One Expert Is Enough For Mixture-of-experts LLMS","date":"2024-11-27","arxiv_id":"2411.18797","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-code-switching-asr-leveraging-non","title":"Enhancing Code-Switching ASR Leveraging Non-Peaky CTC Loss and Deep Language Posterior Injection","date":"2024-11-26","arxiv_id":"2412.08651","repositories_listed":0,"syntology":null},{"url":null,"slug":"ldacp-long-delayed-ad-conversions-prediction","title":"LDACP: Long-Delayed Ad Conversions Prediction Model for Bidding Strategy","date":"2024-11-25","arxiv_id":"2411.16095","repositories_listed":0,"syntology":null},{"url":null,"slug":"mh-moe-multi-head-mixture-of-experts","title":"MH-MoE: Multi-Head Mixture-of-Experts","date":"2024-11-25","arxiv_id":"2411.16205","repositories_listed":0,"syntology":null},{"url":null,"slug":"lifelong-knowledge-editing-for-vision","title":"Lifelong Knowledge Editing for Vision Language Models with Low-Rank Mixture-of-Experts","date":"2024-11-23","arxiv_id":"2411.15432","repositories_listed":0,"syntology":null},{"url":null,"slug":"kaae-numerical-reasoning-for-knowledge-graphs","title":"KAAE: Numerical Reasoning for Knowledge Graphs via Knowledge-aware Attributes Learning","date":"2024-11-20","arxiv_id":"2411.12950","repositories_listed":0,"syntology":null},{"url":null,"slug":"merlot-a-distilled-llm-based-mixture-of","title":"MERLOT: A Distilled LLM-based Mixture-of-Experts Framework for Scalable Encrypted Traffic Classification","date":"2024-11-20","arxiv_id":"2411.13004","repositories_listed":0,"syntology":null},{"url":null,"slug":"ultra-sparse-memory-network","title":"Ultra-Sparse Memory Network","date":"2024-11-19","arxiv_id":"2411.12364","repositories_listed":0,"syntology":null},{"url":null,"slug":"moe-lightning-high-throughput-moe-inference","title":"MoE-Lightning: High-Throughput MoE Inference on Memory-constrained GPUs","date":"2024-11-18","arxiv_id":"2411.11217","repositories_listed":0,"syntology":null},{"url":null,"slug":"lynx-enabling-efficient-moe-inference-through","title":"Lynx: Enabling Efficient MoE Inference through Dynamic Batch-Aware Expert Selection","date":"2024-11-13","arxiv_id":"2411.08982","repositories_listed":0,"syntology":null},{"url":null,"slug":"sparse-upcycling-inference-inefficient","title":"Sparse Upcycling: Inference Inefficient Finetuning","date":"2024-11-13","arxiv_id":"2411.08968","repositories_listed":0,"syntology":null},{"url":null,"slug":"imitation-learning-from-observations-an","title":"Imitation Learning from Observations: An Autoregressive Mixture of Experts Approach","date":"2024-11-12","arxiv_id":"2411.08232","repositories_listed":0,"syntology":null},{"url":null,"slug":"perft-parameter-efficient-routed-fine-tuning","title":"PERFT: Parameter-Efficient Routed Fine-Tuning for Mixture-of-Expert Model","date":"2024-11-12","arxiv_id":"2411.08212","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-vision-mixture-of-experts-for","title":"Towards Vision Mixture of Experts for Wildlife Monitoring on the Edge","date":"2024-11-12","arxiv_id":"2411.07834","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-conditional-expert-selection-network","title":"Adaptive Conditional Expert Selection Network for Multi-domain Recommendation","date":"2024-11-11","arxiv_id":"2411.06826","repositories_listed":0,"syntology":null},{"url":null,"slug":"wdmoe-wireless-distributed-mixture-of-experts","title":"WDMoE: Wireless Distributed Mixture of Experts for Large Language Models","date":"2024-11-11","arxiv_id":"2411.06681","repositories_listed":0,"syntology":null},{"url":null,"slug":"neko-toward-post-recognition-generative","title":"NeKo: Toward Post Recognition Generative Correction Large Language Models with Task-Oriented Experts","date":"2024-11-08","arxiv_id":"2411.05945","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-robust-underwater-acoustic-target","title":"Advancing Robust Underwater Acoustic Target Recognition through Multi-task Learning and Multi-Gate Mixture-of-Experts","date":"2024-11-05","arxiv_id":"2411.02787","repositories_listed":0,"syntology":null},{"url":null,"slug":"fedmoe-da-federated-mixture-of-experts-via","title":"FedMoE-DA: Federated Mixture of Experts via Domain Aware Fine-grained Aggregation","date":"2024-11-04","arxiv_id":"2411.02115","repositories_listed":0,"syntology":null},{"url":null,"slug":"facet-aware-multi-head-mixture-of-experts","title":"Facet-Aware Multi-Head Mixture-of-Experts Model for Sequential Recommendation","date":"2024-11-03","arxiv_id":"2411.01457","repositories_listed":0,"syntology":null},{"url":null,"slug":"hobbit-a-mixed-precision-expert-offloading","title":"HOBBIT: A Mixed Precision Expert Offloading System for Fast MoE Inference","date":"2024-11-03","arxiv_id":"2411.01433","repositories_listed":0,"syntology":null},{"url":null,"slug":"rs-moe-mixture-of-experts-for-remote-sensing","title":"RS-MoE: Mixture of Experts for Remote Sensing Image Captioning and Visual Question Answering","date":"2024-11-03","arxiv_id":"2411.01595","repositories_listed":0,"syntology":null},{"url":null,"slug":"pmol-parameter-efficient-moe-for-preference","title":"PMoL: Parameter Efficient MoE for Preference Mixing of LLM Alignment","date":"2024-11-02","arxiv_id":"2411.01245","repositories_listed":0,"syntology":null},{"url":null,"slug":"moe-i-2-compressing-mixture-of-experts-models","title":"MoE-I$^2$: Compressing Mixture of Experts Models through Inter-Expert Pruning and Intra-Expert Low-Rank Decomposition","date":"2024-11-01","arxiv_id":"2411.01016","repositories_listed":0,"syntology":null},{"url":null,"slug":"stereo-talker-audio-driven-3d-human-synthesis","title":"Stereo-Talker: Audio-driven 3D Human Synthesis with Prior-Guided Mixture-of-Experts","date":"2024-10-31","arxiv_id":"2410.23836","repositories_listed":0,"syntology":null},{"url":null,"slug":"malora-mixture-of-asymmetric-low-rank","title":"MALoRA: Mixture of Asymmetric Low-Rank Adaptation for Enhanced Multi-Task Learning","date":"2024-10-30","arxiv_id":"2410.22782","repositories_listed":0,"syntology":null},{"url":null,"slug":"stealing-user-prompts-from-mixture-of-experts","title":"Stealing User Prompts from Mixture of Experts","date":"2024-10-30","arxiv_id":"2410.22884","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-and-effective-weight-ensembling","title":"Efficient and Effective Weight-Ensembling Mixture of Experts for Multi-Task Model Merging","date":"2024-10-29","arxiv_id":"2410.21804","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-experts-mixture-of-experts-for","title":"Neural Experts: Mixture of Experts for Implicit Neural Representations","date":"2024-10-29","arxiv_id":"2410.21643","repositories_listed":0,"syntology":null},{"url":null,"slug":"promoe-fast-moe-based-llm-serving-using","title":"ProMoE: Fast MoE-based LLM Serving using Proactive Caching","date":"2024-10-29","arxiv_id":"2410.22134","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-mixture-of-expert-for-video-based","title":"Efficient Mixture-of-Expert for Video-based Driver State and Physiological Multi-task Estimation in Conditional Autonomous Driving","date":"2024-10-28","arxiv_id":"2410.21086","repositories_listed":0,"syntology":null},{"url":null,"slug":"finteamexperts-role-specialized-moes-for","title":"FinTeamExperts: Role Specialized MOEs For Financial Analysis","date":"2024-10-28","arxiv_id":"2410.21338","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-of-parrots-experts-improve","title":"Mixture of Parrots: Experts improve memorization more than reasoning","date":"2024-10-24","arxiv_id":"2410.19034","repositories_listed":0,"syntology":null},{"url":null,"slug":"momq-mixture-of-experts-enhances-multi","title":"MoMQ: Mixture-of-Experts Enhances Multi-Dialect Query Generation across Relational and Non-Relational Databases","date":"2024-10-24","arxiv_id":"2410.18406","repositories_listed":0,"syntology":null},{"url":null,"slug":"expertflow-optimized-expert-activation-and","title":"ExpertFlow: Optimized Expert Activation and Token Allocation for Efficient Mixture-of-Experts Inference","date":"2024-10-23","arxiv_id":"2410.17954","repositories_listed":0,"syntology":null},{"url":null,"slug":"faster-language-models-with-better-multi","title":"Faster Language Models with Better Multi-Token Prediction Using Tensor Decomposition","date":"2024-10-23","arxiv_id":"2410.17765","repositories_listed":0,"syntology":null},{"url":null,"slug":"milora-efficient-mixture-of-low-rank","title":"MiLoRA: Efficient Mixture of Low-Rank Adaptation for Large Language Models Fine-tuning","date":"2024-10-23","arxiv_id":"2410.18035","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-and-explainable-depression","title":"Robust and Explainable Depression Identification from Speech Using Vowel-Based Ensemble Learning Approaches","date":"2024-10-23","arxiv_id":"2410.18298","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-mixture-of-experts-inference-time","title":"Optimizing Mixture-of-Experts Inference Time Combining Model Deployment and Communication Scheduling","date":"2024-10-22","arxiv_id":"2410.17043","repositories_listed":0,"syntology":null},{"url":null,"slug":"vimoe-an-empirical-study-of-designing-vision","title":"ViMoE: An Empirical Study of Designing Vision Mixture-of-Experts","date":"2024-10-21","arxiv_id":"2410.15732","repositories_listed":0,"syntology":null},{"url":null,"slug":"mentor-mixture-of-experts-network-with-task","title":"MENTOR: Mixture-of-Experts Network with Task-Oriented Perturbation for Visual Reinforcement Learning","date":"2024-10-19","arxiv_id":"2410.14972","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-generalization-in-sparse-mixture-of","title":"Enhancing Generalization in Sparse Mixture of Experts Models: The Case for Increased Expert Activation in Compositional Tasks","date":"2024-10-17","arxiv_id":"2410.13964","repositories_listed":0,"syntology":null},{"url":null,"slug":"eps-moe-expert-pipeline-scheduler-for-cost","title":"EPS-MoE: Expert Pipeline Scheduler for Cost-Efficient MoE Inference","date":"2024-10-16","arxiv_id":"2410.12247","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-risk-of-evidence-pollution-for","title":"On the Risk of Evidence Pollution for Malicious Social Text Detection in the Era of LLMs","date":"2024-10-16","arxiv_id":"2410.12600","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-expert-structures-on-minimax","title":"Understanding Expert Structures on Minimax Parameter Estimation in Contaminated Mixture of Experts","date":"2024-10-16","arxiv_id":"2410.12258","repositories_listed":0,"syntology":null},{"url":null,"slug":"moe-pruner-pruning-mixture-of-experts-large","title":"MoE-Pruner: Pruning Mixture-of-Experts Large Language Model using the Hints from Its Router","date":"2024-10-15","arxiv_id":"2410.12013","repositories_listed":0,"syntology":null},{"url":null,"slug":"quadratic-gating-functions-in-mixture-of","title":"Quadratic Gating Functions in Mixture of Experts: A Statistical Insight","date":"2024-10-15","arxiv_id":"2410.11222","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformer-layer-injection-a-novel-approach","title":"Transformer Layer Injection: A Novel Approach for Efficient Upscaling of Large Language Models","date":"2024-10-15","arxiv_id":"2410.11654","repositories_listed":0,"syntology":null},{"url":null,"slug":"ada-k-routing-boosting-the-efficiency-of-moe","title":"Ada-K Routing: Boosting the Efficiency of MoE-based LLMs","date":"2024-10-14","arxiv_id":"2410.10456","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-ground-vlms-without-forgetting","title":"Learning to Ground VLMs without Forgetting","date":"2024-10-14","arxiv_id":"2410.10491","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-multi-domain-adaptation-of-language","title":"Scalable Multi-Domain Adaptation of Language Models using Modular Experts","date":"2024-10-14","arxiv_id":"2410.10181","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextwin-whittle-index-based-mixture-of","title":"ContextWIN: Whittle Index Based Mixture-of-Experts Neural Model For Restless Bandits Via Deep RL","date":"2024-10-13","arxiv_id":"2410.09781","repositories_listed":0,"syntology":null},{"url":null,"slug":"moin-mixture-of-introvert-experts-to-upcycle","title":"MoIN: Mixture of Introvert Experts to Upcycle an LLM","date":"2024-10-13","arxiv_id":"2410.09687","repositories_listed":0,"syntology":null},{"url":null,"slug":"at-moe-adaptive-task-planning-mixture-of","title":"AT-MoE: Adaptive Task-planning Mixture of Experts via LoRA Approach","date":"2024-10-12","arxiv_id":"2410.10896","repositories_listed":0,"syntology":null},{"url":"/paper/gets-ensemble-temperature-scaling-for","slug":"gets-ensemble-temperature-scaling-for","title":"GETS: Ensemble Temperature Scaling for Calibration in Graph Neural Networks","date":"2024-10-12","arxiv_id":"2410.09570","repositories_listed":0,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gets-ensemble-temperature-scaling-for#ran","syntology_url":"https://syntology.ai/paper/2410.09570","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.09570"}},"official":null}},{"url":"/paper/mono-internvl-pushing-the-boundaries-of","slug":"mono-internvl-pushing-the-boundaries-of","title":"Mono-InternVL: Pushing the Boundaries of Monolithic Multimodal Large Language Models with Endogenous Visual Pre-training","date":"2024-10-10","arxiv_id":"2410.08202","repositories_listed":0,"syntology":null},{"url":null,"slug":"upcycling-large-language-models-into-mixture","title":"Upcycling Large Language Models into Mixture of Experts","date":"2024-10-10","arxiv_id":"2410.07524","repositories_listed":0,"syntology":null},{"url":null,"slug":"functional-level-uncertainty-quantification","title":"Functional-level Uncertainty Quantification for Calibrated Fine-tuning on LLMs","date":"2024-10-09","arxiv_id":"2410.06431","repositories_listed":0,"syntology":null},{"url":null,"slug":"memory-augmented-transformers-can-implement","title":"Toward generalizable learning of all (linear) first-order methods via memory augmented Transformers","date":"2024-10-08","arxiv_id":"2410.07263","repositories_listed":0,"syntology":null},{"url":null,"slug":"probing-the-robustness-of-theory-of-mind-in","title":"Probing the Robustness of Theory of Mind in Large Language Models","date":"2024-10-08","arxiv_id":"2410.06271","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-laws-across-model-architectures-a","title":"Scaling Laws Across Model Architectures: A Comparative Analysis of Dense and MoE Models in Large Language Models","date":"2024-10-08","arxiv_id":"2410.05661","repositories_listed":0,"syntology":null},{"url":null,"slug":"realizing-video-summarization-from-the-path","title":"Realizing Video Summarization from the Path of Language-based Semantic Understanding","date":"2024-10-06","arxiv_id":"2410.04511","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-dynamic-approach-to-stock-price-prediction","title":"A Dynamic Approach to Stock Price Prediction: Comparing RNN and Mixture of Experts Models Across Different Volatility Profiles","date":"2024-10-04","arxiv_id":"2410.07234","repositories_listed":0,"syntology":null},{"url":null,"slug":"structure-enhanced-protein-instruction-tuning","title":"Structure-Enhanced Protein Instruction Tuning: Towards General-Purpose Protein Understanding with LLMs","date":"2024-10-04","arxiv_id":"2410.03553","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-residual-learning-with-mixture-of","title":"Efficient Residual Learning with Mixture-of-Experts for Universal Dexterous Grasping","date":"2024-10-03","arxiv_id":"2410.02475","repositories_listed":0,"syntology":null},{"url":null,"slug":"neutral-residues-revisiting-adapters-for","title":"Neutral residues: revisiting adapters for model extension","date":"2024-10-03","arxiv_id":"2410.02744","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-expert-estimation-in-hierarchical-mixture","title":"On Expert Estimation in Hierarchical Mixture of Experts: Beyond Softmax Gating Functions","date":"2024-10-03","arxiv_id":"2410.02935","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-prefix-tuning-statistical-benefits","title":"Revisiting Prefix-tuning: Statistical Benefits of Reparameterization among Prompts","date":"2024-10-03","arxiv_id":"2410.02200","repositories_listed":0,"syntology":null},{"url":null,"slug":"ec-dit-scaling-diffusion-transformers-with","title":"EC-DIT: Scaling Diffusion Transformers with Adaptive Expert-Choice Routing","date":"2024-10-02","arxiv_id":"2410.02098","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-labyrinth-of-links-navigating-the","title":"The Labyrinth of Links: Navigating the Associative Maze of Multi-modal LLMs","date":"2024-10-02","arxiv_id":"2410.01417","repositories_listed":0,"syntology":null},{"url":null,"slug":"upcycling-instruction-tuning-from-dense-to","title":"Upcycling Instruction Tuning from Dense to Mixture-of-Experts via Parameter Merging","date":"2024-10-02","arxiv_id":"2410.01610","repositories_listed":0,"syntology":null},{"url":null,"slug":"mos-unleashing-parameter-efficiency-of-low","title":"MoS: Unleashing Parameter Efficiency of Low-Rank Adaptation with Mixture of Shards","date":"2024-10-01","arxiv_id":"2410.00938","repositories_listed":0,"syntology":null},{"url":null,"slug":"uniadapt-a-universal-adapter-for-knowledge","title":"UniAdapt: A Universal Adapter for Knowledge Calibration","date":"2024-10-01","arxiv_id":"2410.00454","repositories_listed":0,"syntology":null},{"url":"/paper/mm1-5-methods-analysis-insights-from","slug":"mm1-5-methods-analysis-insights-from","title":"MM1.5: Methods, Analysis & Insights from Multimodal LLM Fine-tuning","date":"2024-09-30","arxiv_id":"2409.20566","repositories_listed":0,"syntology":null},{"url":null,"slug":"idea-an-inverse-domain-expert-adaptation","title":"IDEA: An Inverse Domain Expert Adaptation Based Active DNN IP Protection Method","date":"2024-09-29","arxiv_id":"2410.00059","repositories_listed":0,"syntology":null},{"url":null,"slug":"scidfm-a-large-language-model-with-mixture-of","title":"SciDFM: A Large Language Model with Mixture-of-Experts for Science","date":"2024-09-27","arxiv_id":"2409.18412","repositories_listed":0,"syntology":null},{"url":null,"slug":"boosting-code-switching-asr-with-mixture-of","title":"Boosting Code-Switching ASR with Mixture of Experts Enhanced Speech-Conditioned LLM","date":"2024-09-24","arxiv_id":"2409.15905","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-mixture-of-experts-for-improved","title":"Leveraging Mixture of Experts for Improved Speech Deepfake Detection","date":"2024-09-24","arxiv_id":"2409.16077","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-mixture-of-experts-enabled-trustworthy","title":"Toward Mixture-of-Experts Enabled Trustworthy Semantic Communication for 6G Networks","date":"2024-09-24","arxiv_id":"2409.15695","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-modal-generative-ai-multi-modal-llm","title":"Multi-Modal Generative AI: Multi-modal LLM, Diffusion and Beyond","date":"2024-09-23","arxiv_id":"2409.14993","repositories_listed":0,"syntology":null},{"url":null,"slug":"2409-14107","title":"Routing in Sparsely-gated Language Models responds to Context","date":"2024-09-21","arxiv_id":"2409.14107","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-omics-data-integration-for-early","title":"Multi-omics data integration for early diagnosis of hepatocellular carcinoma (HCC) using machine learning","date":"2024-09-20","arxiv_id":"2409.13791","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-audiovisual-speech-recognition-models","title":"Robust Audiovisual Speech Recognition Models with Mixture-of-Experts","date":"2024-09-19","arxiv_id":"2409.12370","repositories_listed":0,"syntology":null},{"url":null,"slug":"grin-gradient-informed-moe","title":"GRIN: GRadient-INformed MoE","date":"2024-09-18","arxiv_id":"2409.12136","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-of-diverse-size-experts","title":"Mixture of Diverse Size Experts","date":"2024-09-18","arxiv_id":"2409.12210","repositories_listed":0,"syntology":null},{"url":null,"slug":"lpt-efficient-training-on-mixture-of-long","title":"LPT++: Efficient Training on Mixture of Long-tailed Experts","date":"2024-09-17","arxiv_id":"2409.11323","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-segmentation-based-initialization","title":"Adaptive Segmentation-Based Initialization for Steered Mixture of Experts Image Regression","date":"2024-09-16","arxiv_id":"2409.10101","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-ai-s-carbon-footprint-into-risk","title":"Integrating AI's Carbon Footprint into Risk Management Frameworks: Strategies and Tools for Sustainable Compliance in Banking Sector","date":"2024-09-15","arxiv_id":"2410.01818","repositories_listed":0,"syntology":null},{"url":null,"slug":"da-moe-towards-dynamic-expert-allocation-for","title":"DA-MoE: Towards Dynamic Expert Allocation for Mixture-of-Experts Models","date":"2024-09-10","arxiv_id":"2409.06669","repositories_listed":0,"syntology":null},{"url":null,"slug":"stun-structured-then-unstructured-pruning-for","title":"STUN: Structured-Then-Unstructured Pruning for Scalable MoE Pruning","date":"2024-09-10","arxiv_id":"2409.06211","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapted-moe-mixture-of-experts-with-test-time","title":"Adapted-MoE: Mixture of Experts with Test-Time Adaption for Anomaly Detection","date":"2024-09-09","arxiv_id":"2409.05611","repositories_listed":0,"syntology":null}],"record_sha256":"fedff034713c6d008fad9633e05183513df38b421ce68adea14decfb66bf7ed6","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}