{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/mixture-of-experts/papers/12","list_of":"/task/mixture-of-experts","task":"Mixture-of-Experts","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":12,"pages_in_order":14,"rows_per_page":100,"rows":[1101,1200],"of":1312,"counts":{"archive_papers_tagged":1312,"with_a_code_link":516,"where_syntology_ran_a_sample":216,"not_listed_spam_title":0,"listed":1312,"listed_where_code_ran":216,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":184,"every_run_a_failure_of_syntologys_instrument":32,"listed_with_a_run_with_no_instrument_failure":184,"listed_every_run_a_failure_of_syntologys_instrument":32,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/mixture-of-experts","prev":"/task/mixture-of-experts/papers/11","next":"/task/mixture-of-experts/papers/13","papers":[{"url":null,"slug":"towards-moe-deployment-mitigating","title":"Towards MoE Deployment: Mitigating Inefficiencies in Mixture-of-Expert (MoE) Inference","date":"2023-03-10","arxiv_id":"2303.06182","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-expert-specialization-in-mixture-of","title":"Improving Expert Specialization in Mixture of Experts","date":"2023-02-28","arxiv_id":"2302.14703","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-training-of-mixture-of-experts","title":"Improved Training of Mixture-of-Experts Language GANs","date":"2023-02-23","arxiv_id":"2302.11875","repositories_listed":0,"syntology":null},{"url":null,"slug":"tmoe-p-towards-the-pareto-optimum-for","title":"TMoE-P: Towards the Pareto Optimum for Multivariate Soft Sensors","date":"2023-02-21","arxiv_id":"2302.10477","repositories_listed":0,"syntology":null},{"url":null,"slug":"massively-multilingual-shallow-fusion-with","title":"Massively Multilingual Shallow Fusion with Large Language Models","date":"2023-02-17","arxiv_id":"2302.08917","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-differentiable-and-sparse-top-k-a-convex","title":"Fast, Differentiable and Sparse Top-k: a Convex Analysis Perspective","date":"2023-02-02","arxiv_id":"2302.01425","repositories_listed":0,"syntology":null},{"url":null,"slug":"alternating-updates-for-efficient","title":"Alternating Updates for Efficient Transformers","date":"2023-01-30","arxiv_id":"2301.13310","repositories_listed":0,"syntology":null},{"url":null,"slug":"prudex-compass-towards-systematic-evaluation","title":"PRUDEX-Compass: Towards Systematic Evaluation of Reinforcement Learning in Financial Markets","date":"2023-01-14","arxiv_id":"2302.00586","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaensemble-learning-adaptively-sparse","title":"AdaEnsemble: Learning Adaptively Sparse Structured Ensemble Network for Click-Through Rate Prediction","date":"2023-01-06","arxiv_id":"2301.08353","repositories_listed":0,"syntology":null},{"url":null,"slug":"mod-squad-designing-mixtures-of-experts-as","title":"Mod-Squad: Designing Mixtures of Experts As Modular Multi-Task Learners","date":"2023-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-aware-dynamic-parameter-for-video","title":"Semantic-Aware Dynamic Parameter for Video Inpainting Transformer","date":"2023-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"generalizing-multimodal-variational-methods","title":"Generalizing Multimodal Variational Methods to Sets","date":"2022-12-19","arxiv_id":"2212.09918","repositories_listed":0,"syntology":null},{"url":null,"slug":"memory-efficient-nllb-200-language-specific","title":"Memory-efficient NLLB-200: Language-specific Expert Pruning of a Massively Multilingual Machine Translation Model","date":"2022-12-19","arxiv_id":"2212.09811","repositories_listed":0,"syntology":null},{"url":null,"slug":"multicoder-multi-programming-lingual-pre","title":"MultiCoder: Multi-Programming-Lingual Pre-Training for Low-Resource Code Completion","date":"2022-12-19","arxiv_id":"2212.09666","repositories_listed":0,"syntology":null},{"url":null,"slug":"fixing-moe-over-fitting-on-low-resource","title":"Fixing MoE Over-Fitting on Low-Resource Languages in Multilingual Machine Translation","date":"2022-12-15","arxiv_id":"2212.07571","repositories_listed":0,"syntology":null},{"url":null,"slug":"mod-squad-designing-mixture-of-experts-as","title":"Mod-Squad: Designing Mixture of Experts As Modular Multi-Task Learners","date":"2022-12-15","arxiv_id":"2212.08066","repositories_listed":0,"syntology":null},{"url":null,"slug":"smile-scaling-mixture-of-experts-with","title":"SMILE: Scaling Mixture-of-Experts with Efficient Bi-level Routing","date":"2022-12-10","arxiv_id":"2212.05191","repositories_listed":0,"syntology":null},{"url":null,"slug":"incorporating-polar-field-data-for-improved","title":"Incorporating Polar Field Data for Improved Solar Flare Prediction","date":"2022-12-04","arxiv_id":"2212.01730","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatically-extracting-information-in","title":"Automatically Extracting Information in Medical Dialogue: Expert System And Attention for Labelling","date":"2022-11-28","arxiv_id":"2211.15544","repositories_listed":0,"syntology":null},{"url":null,"slug":"double-deep-q-learning-in-opponent-modeling","title":"Double Deep Q-Learning in Opponent Modeling","date":"2022-11-24","arxiv_id":"2211.15384","repositories_listed":0,"syntology":null},{"url":null,"slug":"who-says-elephants-can-t-run-bringing-large","title":"Who Says Elephants Can't Run: Bringing Large Scale MoE Models into Cloud Scale Production","date":"2022-11-18","arxiv_id":"2211.10017","repositories_listed":0,"syntology":null},{"url":null,"slug":"hmoe-hypernetwork-based-mixture-of-experts","title":"HMOE: Hypernetwork-based Mixture of Experts for Domain Generalization","date":"2022-11-15","arxiv_id":"2211.08253","repositories_listed":0,"syntology":null},{"url":null,"slug":"breaking-trade-offs-in-speech-separation-with","title":"Handling Trade-Offs in Speech Separation with Sparsely-Gated Mixture of Experts","date":"2022-11-11","arxiv_id":"2211.06493","repositories_listed":0,"syntology":null},{"url":null,"slug":"speechmatrix-a-large-scale-mined-corpus-of-1","title":"SpeechMatrix: A Large-Scale Mined Corpus of Multilingual Speech-to-Speech Translations","date":"2022-11-08","arxiv_id":"2211.04508","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-deep-mixture-of-experts-to-detect-word","title":"Using Deep Mixture-of-Experts to Detect Word Meaning Shift for TempoWiC","date":"2022-11-07","arxiv_id":"2211.03466","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-real-world-autonomous-driving-by","title":"Safe Real-World Autonomous Driving by Learning to Predict and Plan with a Mixture of Experts","date":"2022-11-03","arxiv_id":"2211.02131","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-mixture-of-experts-integrating","title":"Contextual Mixture of Experts: Integrating Knowledge into Predictive Modeling","date":"2022-11-01","arxiv_id":"2211.00558","repositories_listed":0,"syntology":null},{"url":null,"slug":"prediction-sets-for-high-dimensional-mixture","title":"Prediction Sets for High-Dimensional Mixture of Experts Models","date":"2022-10-30","arxiv_id":"2210.16710","repositories_listed":0,"syntology":null},{"url":"/paper/knowledge-in-context-towards-knowledgeable","slug":"knowledge-in-context-towards-knowledgeable","title":"Knowledge-in-Context: Towards Knowledgeable Semi-Parametric Language Models","date":"2022-10-28","arxiv_id":"2210.16433","repositories_listed":0,"syntology":null},{"url":null,"slug":"coordination-with-humans-via-strategy","title":"Coordination with Humans via Strategy Matching","date":"2022-10-27","arxiv_id":"2210.15099","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-adversarial-robustness-of-mixture-of","title":"On the Adversarial Robustness of Mixture of Experts","date":"2022-10-19","arxiv_id":"2210.10253","repositories_listed":0,"syntology":null},{"url":null,"slug":"tiny-attention-adapter-contexts-are-more","title":"Tiny-Attention Adapter: Contexts Are More Important Than the Number of Parameters","date":"2022-10-18","arxiv_id":"2211.01979","repositories_listed":0,"syntology":null},{"url":null,"slug":"feamoe-fair-explainable-and-adaptive-mixture","title":"FEAMOE: Fair, Explainable and Adaptive Mixture of Experts","date":"2022-10-10","arxiv_id":"2210.04995","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-mixture-of-experts-approach-for","title":"Deep Learning Mixture-of-Experts Approach for Cytotoxic Edema Assessment in Infants and Children","date":"2022-10-06","arxiv_id":"2210.04767","repositories_listed":0,"syntology":null},{"url":null,"slug":"probabilistic-partition-of-unity-networks-for","title":"Probabilistic partition of unity networks for high-dimensional regression problems","date":"2022-10-06","arxiv_id":"2210.02694","repositories_listed":0,"syntology":null},{"url":null,"slug":"parameter-varying-neural-ordinary","title":"Parameter-varying neural ordinary differential equations with partition-of-unity networks","date":"2022-10-01","arxiv_id":"2210.00368","repositories_listed":0,"syntology":null},{"url":null,"slug":"table-based-fact-verification-with-self-2","title":"Table-based Fact Verification with Self-labeled Keypoint Alignment","date":"2022-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-of-experts-models-for-multilevel-data","title":"Mixture of experts models for multilevel data: modelling framework and approximation theory","date":"2022-09-30","arxiv_id":"2209.15207","repositories_listed":0,"syntology":null},{"url":null,"slug":"sparsity-constrained-optimal-transport","title":"Sparsity-Constrained Optimal Transport","date":"2022-09-30","arxiv_id":"2209.15466","repositories_listed":0,"syntology":null},{"url":null,"slug":"tuning-of-mixture-of-experts-mixed-precision","title":"Tuning of Mixture-of-Experts Mixed-Precision Neural Networks","date":"2022-09-29","arxiv_id":"2209.15427","repositories_listed":0,"syntology":null},{"url":null,"slug":"diversified-dynamic-routing-for-vision-tasks","title":"Diversified Dynamic Routing for Vision Tasks","date":"2022-09-26","arxiv_id":"2209.13071","repositories_listed":0,"syntology":null},{"url":null,"slug":"parameter-efficient-conformers-via-sharing","title":"Parameter-Efficient Conformers via Sharing Sparsely-Gated Experts for End-to-End Speech Recognition","date":"2022-09-17","arxiv_id":"2209.08326","repositories_listed":0,"syntology":null},{"url":null,"slug":"sparse-video-representation-using-steered","title":"Sparse Video Representation Using Steered Mixture-of-Experts With Global Motion Compensation","date":"2022-09-13","arxiv_id":"2209.05993","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-sparse-expert-models-in-deep","title":"A Review of Sparse Expert Models in Deep Learning","date":"2022-09-04","arxiv_id":"2209.01667","repositories_listed":0,"syntology":null},{"url":null,"slug":"came-context-aware-mixture-of-experts-for","title":"Context-aware Mixture-of-Experts for Unbiased Scene Graph Generation","date":"2022-08-15","arxiv_id":"2208.07109","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-theoretical-view-on-sparsely-activated","title":"A Theoretical View on Sparsely Activated Networks","date":"2022-08-08","arxiv_id":"2208.04461","repositories_listed":0,"syntology":null},{"url":null,"slug":"edge-aware-autoencoder-design-for-real-time","title":"Edge-Aware Autoencoder Design for Real-Time Mixture-of-Experts Image Compression","date":"2022-07-25","arxiv_id":"2207.12348","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-mixture-of-experts-learning-for","title":"Adaptive Mixture of Experts Learning for Generalizable Face Anti-Spoofing","date":"2022-07-20","arxiv_id":"2207.09868","repositories_listed":0,"syntology":null},{"url":null,"slug":"moec-mixture-of-expert-clusters","title":"MoEC: Mixture of Expert Clusters","date":"2022-07-19","arxiv_id":"2207.09094","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-large-scale-universal-user","title":"Learning Large-scale Universal User Representation with Sparse Mixture of Experts","date":"2022-07-11","arxiv_id":"2207.04648","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-neural-data-server-a-data-1","title":"Scalable Neural Data Server: A Data Recommender for Transfer Learning","date":"2022-06-19","arxiv_id":"2206.09386","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantitative-stock-investment-by-routing","title":"Quantitative Stock Investment by Routing Uncertainty-Aware Trading Experts: A Multi-Task Learning Approach","date":"2022-06-07","arxiv_id":"2207.07578","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-contrastive-learning-with-limoe","title":"Multimodal Contrastive Learning with LIMoE: the Language-Image Mixture of Experts","date":"2022-06-06","arxiv_id":"2206.02770","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-mixture-of-experts-for","title":"Interpretable Mixture of Experts","date":"2022-06-05","arxiv_id":"2206.02107","repositories_listed":0,"syntology":null},{"url":null,"slug":"task-specific-expert-pruning-for-sparse","title":"Task-Specific Expert Pruning for Sparse Mixture-of-Experts","date":"2022-06-01","arxiv_id":"2206.00277","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-expert-selection-for-multi-scenario","title":"Automatic Expert Selection for Multi-Scenario and Multi-Task Search","date":"2022-05-28","arxiv_id":"2205.14321","repositories_listed":0,"syntology":null},{"url":null,"slug":"gating-dropout-communication-efficient","title":"Gating Dropout: Communication-efficient Regularization for Sparsely Activated Transformers","date":"2022-05-28","arxiv_id":"2205.14336","repositories_listed":0,"syntology":null},{"url":null,"slug":"pluralistic-image-completion-with","title":"Pluralistic Image Completion with Probabilistic Mixture-of-Experts","date":"2022-05-18","arxiv_id":"2205.09086","repositories_listed":0,"syntology":null},{"url":null,"slug":"unified-modeling-of-multi-domain-multi-device","title":"Unified Modeling of Multi-Domain Multi-Device ASR Systems","date":"2022-05-13","arxiv_id":"2205.06655","repositories_listed":0,"syntology":null},{"url":null,"slug":"st-expertnet-a-deep-expert-framework-for","title":"ST-ExpertNet: A Deep Expert Framework for Traffic Prediction","date":"2022-05-05","arxiv_id":"2205.07851","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-mixture-of-experts-using-dynamic","title":"Optimizing Mixture of Experts using Dynamic Recompilations","date":"2022-05-04","arxiv_id":"2205.01848","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-can-cross-lingual-knowledge-contribute","title":"How Can Cross-lingual Knowledge Contribute Better to Fine-Grained Entity Typing?","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"residual-mixture-of-experts","title":"Residual Mixture of Experts","date":"2022-04-20","arxiv_id":"2204.09636","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-efficient-single-image-dehazing-and","title":"Towards Efficient Single Image Dehazing and Desnowing","date":"2022-04-19","arxiv_id":"2204.08899","repositories_listed":0,"syntology":null},{"url":null,"slug":"sparsely-activated-mixture-of-experts-are-1","title":"Sparsely Activated Mixture-of-Experts are Robust Multi-Task Learners","date":"2022-04-16","arxiv_id":"2204.07689","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-of-experts-for-biomedical-question","title":"Mixture of Experts for Biomedical Question Answering","date":"2022-04-15","arxiv_id":"2204.07469","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-of-experts-vaes-can-disregard","title":"Mixture-of-experts VAEs can disregard variation in surjective multimodal data","date":"2022-04-11","arxiv_id":"2204.05229","repositories_listed":0,"syntology":null},{"url":null,"slug":"concept-drift-adaptation-for-ctr-prediction","title":"On the Adaptation to Concept Drift for CTR Prediction","date":"2022-04-01","arxiv_id":"2204.05101","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-reflectance-capture-with-a-deep","title":"Efficient Reflectance Capture with a Deep Gated Mixture-of-Experts","date":"2022-03-29","arxiv_id":"2203.15258","repositories_listed":0,"syntology":null},{"url":"/paper/efficient-language-modeling-with-sparse-all","slug":"efficient-language-modeling-with-sparse-all","title":"Efficient Language Modeling with Sparse all-MLP","date":"2022-03-14","arxiv_id":"2203.06850","repositories_listed":0,"syntology":null},{"url":null,"slug":"one-model-multiple-tasks-pathways-for-natural","title":"SkillNet-NLU: A Sparsely Activated Model for General-Purpose Natural Language Understanding","date":"2022-03-07","arxiv_id":"2203.03312","repositories_listed":0,"syntology":null},{"url":null,"slug":"functional-mixture-of-experts-for","title":"Functional mixture-of-experts for classification","date":"2022-02-28","arxiv_id":"2202.13934","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-of-experts-with-expert-choice-routing","title":"Mixture-of-Experts with Expert Choice Routing","date":"2022-02-18","arxiv_id":"2202.09368","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-dynamic-neural-networks-for","title":"A Survey on Dynamic Neural Networks for Natural Language Processing","date":"2022-02-15","arxiv_id":"2202.07101","repositories_listed":0,"syntology":null},{"url":null,"slug":"physics-guided-problem-decomposition-for","title":"Physics-Guided Problem Decomposition for Scaling Deep Learning of High-dimensional Eigen-Solvers: The Case of Schrödinger's Equation","date":"2022-02-12","arxiv_id":"2202.05994","repositories_listed":0,"syntology":null},{"url":null,"slug":"one-student-knows-all-experts-know-from","title":"One Student Knows All Experts Know: From Sparse to Dense","date":"2022-01-26","arxiv_id":"2201.10890","repositories_listed":0,"syntology":null},{"url":null,"slug":"moebert-from-bert-to-mixture-of-experts-via","title":"MoEBERT: from BERT to Mixture-of-Experts via Importance-Guided Adaptation","date":"2022-01-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sparsely-activated-mixture-of-experts-are","title":"Sparsely Activated Mixture-of-Experts are Robust Multi-Task Learners","date":"2022-01-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-lightweight-neural-animation","title":"Towards Lightweight Neural Animation : Exploration of Neural Network Pruning in Mixture of Experts-based Animation Models","date":"2022-01-11","arxiv_id":"2201.04042","repositories_listed":0,"syntology":null},{"url":null,"slug":"combinations-of-adaptive-filters","title":"Combinations of Adaptive Filters","date":"2021-12-22","arxiv_id":"2112.12245","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-large-scale-language-modeling-with","title":"Efficient Large Scale Language Modeling with Mixtures of Experts","date":"2021-12-20","arxiv_id":"2112.10684","repositories_listed":0,"syntology":null},{"url":"/paper/glam-efficient-scaling-of-language-models","slug":"glam-efficient-scaling-of-language-models","title":"GLaM: Efficient Scaling of Language Models with Mixture-of-Experts","date":"2021-12-13","arxiv_id":"2112.06905","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-a-great-multi-lingual-teacher-with","title":"Building a great multi-lingual teacher with sparsely-gated mixture of experts for speech recognition","date":"2021-12-10","arxiv_id":"2112.05820","repositories_listed":0,"syntology":null},{"url":null,"slug":"anchoring-to-exemplars-for-training-mixture","title":"Anchoring to Exemplars for Training Mixture-of-Expert Cell Embeddings","date":"2021-12-06","arxiv_id":"2112.03208","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-mixture-of-expert-based-deep-neural-network","title":"A Mixture of Expert Based Deep Neural Network for Improved ASR","date":"2021-12-02","arxiv_id":"2112.01025","repositories_listed":0,"syntology":null},{"url":null,"slug":"tal-two-stream-adaptive-learning-for","title":"TAL: Two-stream Adaptive Learning for Generalizable Person Re-identification","date":"2021-11-29","arxiv_id":"2111.14290","repositories_listed":0,"syntology":null},{"url":null,"slug":"expert-aggregation-for-financial-forecasting","title":"Expert Aggregation for Financial Forecasting","date":"2021-11-25","arxiv_id":"2111.15365","repositories_listed":0,"syntology":null},{"url":null,"slug":"speechmoe2-mixture-of-experts-model-with","title":"SpeechMoE2: Mixture-of-Experts Model with Improved Routing","date":"2021-11-23","arxiv_id":"2111.11831","repositories_listed":0,"syntology":null},{"url":null,"slug":"m6-t-exploring-sparse-expert-models-and","title":"M6-T: Exploring Sparse Expert Models and Beyond","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"moefication-conditional-computation-of-1","title":"MoEfication: Conditional Computation of Transformer Models for Efficient Inference","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"stablemoe-stable-routing-strategy-for-mixture","title":"StableMoE: Stable Routing Strategy for Mixture of Experts","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"summareranker-a-multi-task-mixture-of-experts","title":"SummaReranker: A Multi-Task Mixture-of-Experts Re-ranking Framework for Abstractive Summarization","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"table-based-fact-verification-with-self","title":"Table-based Fact Verification with Self-adaptive Mixture of Experts","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"rtm-super-learner-results-at-quality","title":"RTM Super Learner Results at Quality Estimation Task","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"polynomial-spline-neural-networks-with-exact","title":"Polynomial-Spline Neural Networks with Exact Integrals","date":"2021-10-26","arxiv_id":"2110.14055","repositories_listed":0,"syntology":null},{"url":null,"slug":"simple-or-complex-complexity-controllable","title":"Simple or Complex? Complexity-Controllable Question Generation with Soft Templates and Deep Mixture of Experts Model","date":"2021-10-13","arxiv_id":"2110.06560","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-learning-using-task-conditional-1","title":"Continual Learning Using Task Conditional Neural Networks","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"full-precision-free-binary-graph-neural","title":"Full-Precision Free Binary Graph Neural Networks","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"hydrasum-disentangling-stylistic-features-in","title":"HydraSum - Disentangling Stylistic Features in Text Summarization using Multi-Decoder Models","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"mecats-mixture-of-experts-for-probabilistic","title":"MECATS: Mixture-of-Experts for Probabilistic Forecasts of Aggregated Time Series","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null}],"record_sha256":"5acc52b61f452c3c9506e51f3f6c77567a03ae72a69c87dae08f36f0fdd905d5","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}