{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/mixture-of-experts/papers/5","list_of":"/task/mixture-of-experts","task":"Mixture-of-Experts","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":5,"pages_in_order":14,"rows_per_page":100,"rows":[401,500],"of":1312,"counts":{"archive_papers_tagged":1312,"with_a_code_link":516,"where_syntology_ran_a_sample":216,"not_listed_spam_title":0,"listed":1312,"listed_where_code_ran":216,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":184,"every_run_a_failure_of_syntologys_instrument":32,"listed_with_a_run_with_no_instrument_failure":184,"listed_every_run_a_failure_of_syntologys_instrument":32,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/mixture-of-experts","prev":"/task/mixture-of-experts/papers/4","next":"/task/mixture-of-experts/papers/6","papers":[{"url":"/paper/taskexpert-dynamically-assembling-multi-task","slug":"taskexpert-dynamically-assembling-multi-task","title":"TaskExpert: Dynamically Assembling Multi-Task Representations with Memorial Mixture-of-Experts","date":"2023-07-28","arxiv_id":"2307.15324","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":1,"n_ran_checked":1,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/taskexpert-dynamically-assembling-multi-task#ran","syntology_url":"https://syntology.ai/paper/2307.15324","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.15324"}},"official":{"repos":["prismformore/multi-task-transformer"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/ntk-approximating-mlp-fusion-for-efficient","slug":"ntk-approximating-mlp-fusion-for-efficient","title":"MLP Fusion: Towards Efficient Fine-tuning of Dense and Mixture-of-Experts Language Models","date":"2023-07-18","arxiv_id":"2307.08941","repositories_listed":1,"syntology":{"n":14,"n_ran":10,"n_constructed":5,"n_ran_checked":9,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":14,"phrase":"10 ran (of which 5 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/ntk-approximating-mlp-fusion-for-efficient#ran","syntology_url":"https://syntology.ai/paper/2307.08941","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.08941"}},"official":{"repos":["weitianxin/mlp_fusion"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":5,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/domain-agnostic-neural-architecture-for-class","slug":"domain-agnostic-neural-architecture-for-class","title":"Domain-Agnostic Neural Architecture for Class Incremental Continual Learning in Document Processing Platform","date":"2023-07-11","arxiv_id":"2307.05399","repositories_listed":1,"syntology":null},{"url":"/paper/bidirectional-attention-as-a-mixture-of","slug":"bidirectional-attention-as-a-mixture-of","title":"Bidirectional Attention as a Mixture of Continuous Word Experts","date":"2023-07-08","arxiv_id":"2307.04057","repositories_listed":1,"syntology":null},{"url":"/paper/chatlaw-open-source-legal-large-language","slug":"chatlaw-open-source-legal-large-language","title":"Chatlaw: A Multi-Agent Collaborative Legal Assistant with Knowledge Graph Enhanced Mixture-of-Experts Large Language Model","date":"2023-06-28","arxiv_id":"2306.16092","repositories_listed":1,"syntology":null},{"url":"/paper/shiftaddvit-mixture-of-multiplication-1","slug":"shiftaddvit-mixture-of-multiplication-1","title":"ShiftAddViT: Mixture of Multiplication Primitives Towards Efficient Vision Transformer","date":"2023-06-10","arxiv_id":"2306.06446","repositories_listed":1,"syntology":{"n":20,"n_ran":14,"n_constructed":4,"n_ran_checked":12,"n_instrument":2,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":2,"phrase":"14 ran (of which 4 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/shiftaddvit-mixture-of-multiplication-1#ran","syntology_url":"https://syntology.ai/paper/2306.06446","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.06446"}},"official":{"repos":["gatech-eic/shiftaddvit"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":4,"n_ran_no_instrument_failure":12,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/mixture-of-supernets-improving-weight-sharing","slug":"mixture-of-supernets-improving-weight-sharing","title":"Mixture-of-Supernets: Improving Weight-Sharing Supernet Training with Architecture-Routed Mixture-of-Experts","date":"2023-06-08","arxiv_id":"2306.04845","repositories_listed":1,"syntology":null},{"url":"/paper/moduleformer-learning-modular-large-language","slug":"moduleformer-learning-modular-large-language","title":"ModuleFormer: Modularity Emerges from Mixture-of-Experts","date":"2023-06-07","arxiv_id":"2306.04640","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/moduleformer-learning-modular-large-language#ran","syntology_url":"https://syntology.ai/paper/2306.04640","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.04640"}},"official":{"repos":["ibm/moduleformer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/revisiting-hate-speech-benchmarks-from-data","slug":"revisiting-hate-speech-benchmarks-from-data","title":"Revisiting Hate Speech Benchmarks: From Data Curation to System Deployment","date":"2023-06-01","arxiv_id":"2306.01105","repositories_listed":1,"syntology":null},{"url":"/paper/edge-moe-memory-efficient-multi-task-vision","slug":"edge-moe-memory-efficient-multi-task-vision","title":"Edge-MoE: Memory-Efficient Multi-Task Vision Transformer Architecture with Task-level Sparsity via Mixture-of-Experts","date":"2023-05-30","arxiv_id":"2305.18691","repositories_listed":1,"syntology":null},{"url":"/paper/raphael-text-to-image-generation-via-large","slug":"raphael-text-to-image-generation-via-large","title":"RAPHAEL: Text-to-Image Generation via Large Mixture of Diffusion Paths","date":"2023-05-29","arxiv_id":"2305.18295","repositories_listed":1,"syntology":null},{"url":"/paper/emergent-modularity-in-pre-trained","slug":"emergent-modularity-in-pre-trained","title":"Emergent Modularity in Pre-trained Transformers","date":"2023-05-28","arxiv_id":"2305.18390","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/emergent-modularity-in-pre-trained#ran","syntology_url":"https://syntology.ai/paper/2305.18390","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18390"}},"official":{"repos":["thunlp/modularity-analysis"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/condensing-multilingual-knowledge-with","slug":"condensing-multilingual-knowledge-with","title":"Condensing Multilingual Knowledge with Lightweight Language-Specific Modules","date":"2023-05-23","arxiv_id":"2305.13993","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/condensing-multilingual-knowledge-with#ran","syntology_url":"https://syntology.ai/paper/2305.13993","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13993"}},"official":{"repos":["fe1ixxu/lms_fd"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lifting-the-curse-of-capacity-gap-in","slug":"lifting-the-curse-of-capacity-gap-in","title":"Lifting the Curse of Capacity Gap in Distilling Language Models","date":"2023-05-20","arxiv_id":"2305.12129","repositories_listed":1,"syntology":{"n":13,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/lifting-the-curse-of-capacity-gap-in#ran","syntology_url":"https://syntology.ai/paper/2305.12129","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.12129"}},"official":{"repos":["genezc/minimoe"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-being-parameter-efficient-a","slug":"towards-being-parameter-efficient-a","title":"Towards Being Parameter-Efficient: A Stratified Sparsely Activated Transformer with Dynamic Capacity","date":"2023-05-03","arxiv_id":"2305.02176","repositories_listed":1,"syntology":null},{"url":"/paper/unicorn-a-unified-multi-tasking-model-for","slug":"unicorn-a-unified-multi-tasking-model-for","title":"Unicorn: A Unified Multi-tasking Model for Supporting Matching Tasks in Data Integration","date":"2023-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/information-maximizing-curriculum-a","slug":"information-maximizing-curriculum-a","title":"Information Maximizing Curriculum: A Curriculum-Based Approach for Imitating Diverse Skills","date":"2023-03-27","arxiv_id":"2303.15349","repositories_listed":1,"syntology":null},{"url":"/paper/learning-a-sparse-transformer-network-for","slug":"learning-a-sparse-transformer-network-for","title":"Learning A Sparse Transformer Network for Effective Image Deraining","date":"2023-03-21","arxiv_id":"2303.11950","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":8,"n_ran_checked":8,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":11,"phrase":"8 ran (of which 8 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified; every one of the 8 samples that ran constructed an object rather than computing a result","sample_list":"/paper/learning-a-sparse-transformer-network-for#ran","syntology_url":"https://syntology.ai/paper/2303.11950","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.11950"}},"official":{"repos":["cschenxiang/drsformer"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":8,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/a-novel-tensor-expert-hybrid-parallelism","slug":"a-novel-tensor-expert-hybrid-parallelism","title":"A Hybrid Tensor-Expert-Data Parallelism Approach to Optimize Mixture-of-Experts Training","date":"2023-03-11","arxiv_id":"2303.06318","repositories_listed":1,"syntology":null},{"url":"/paper/mixphm-redundancy-aware-parameter-efficient","slug":"mixphm-redundancy-aware-parameter-efficient","title":"MixPHM: Redundancy-Aware Parameter-Efficient Tuning for Low-Resource Visual Question Answering","date":"2023-03-02","arxiv_id":"2303.01239","repositories_listed":1,"syntology":null},{"url":"/paper/sparse-moe-as-the-new-dropout-scaling-dense","slug":"sparse-moe-as-the-new-dropout-scaling-dense","title":"Sparse MoE as the New Dropout: Scaling Dense and Self-Slimmable Transformers","date":"2023-03-02","arxiv_id":"2303.01610","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sparse-moe-as-the-new-dropout-scaling-dense#ran","syntology_url":"https://syntology.ai/paper/2303.01610","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.01610"}},"official":{"repos":["vita-group/random-moe-as-dropout"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/covariate-guided-bayesian-mixture-model-for","slug":"covariate-guided-bayesian-mixture-model-for","title":"Covariate-guided Bayesian mixture model for multivariate time series","date":"2023-01-03","arxiv_id":"2301.01373","repositories_listed":1,"syntology":null},{"url":"/paper/adamv-moe-adaptive-multi-task-vision-mixture","slug":"adamv-moe-adaptive-multi-task-vision-mixture","title":"AdaMV-MoE: Adaptive Multi-Task Vision Mixture-of-Experts","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/sparse-upcycling-training-mixture-of-experts","slug":"sparse-upcycling-training-mixture-of-experts","title":"Sparse Upcycling: Training Mixture-of-Experts from Dense Checkpoints","date":"2022-12-09","arxiv_id":"2212.05055","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sparse-upcycling-training-mixture-of-experts#ran","syntology_url":"https://syntology.ai/paper/2212.05055","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.05055"}},"official":{"repos":["google-research/vmoe"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/named-entity-and-relation-extraction-with","slug":"named-entity-and-relation-extraction-with","title":"Named Entity and Relation Extraction with Multi-Modal Retrieval","date":"2022-12-03","arxiv_id":"2212.01612","repositories_listed":1,"syntology":null},{"url":"/paper/mixture-of-decision-trees-for-interpretable","slug":"mixture-of-decision-trees-for-interpretable","title":"Mixture of Decision Trees for Interpretable Machine Learning","date":"2022-11-26","arxiv_id":"2211.14617","repositories_listed":1,"syntology":null},{"url":"/paper/spatial-mixture-of-experts","slug":"spatial-mixture-of-experts","title":"Spatial Mixture-of-Experts","date":"2022-11-24","arxiv_id":"2211.13491","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":2,"n_ran_checked":2,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/spatial-mixture-of-experts#ran","syntology_url":"https://syntology.ai/paper/2211.13491","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.13491"}},"official":{"repos":["spcl/smoe"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-bird-s-eye-view-of-reranking-from-list","slug":"a-bird-s-eye-view-of-reranking-from-list","title":"A Bird's-eye View of Reranking: from List Level to Page Level","date":"2022-11-17","arxiv_id":"2211.09303","repositories_listed":1,"syntology":null},{"url":"/paper/cherry-hypothesis-identifying-the-cherry-on","slug":"cherry-hypothesis-identifying-the-cherry-on","title":"PAD-Net: An Efficient Framework for Dynamic Networks","date":"2022-11-10","arxiv_id":"2211.05528","repositories_listed":1,"syntology":null},{"url":"/paper/m-3-vit-mixture-of-experts-vision-transformer","slug":"m-3-vit-mixture-of-experts-vision-transformer","title":"M$^3$ViT: Mixture-of-Experts Vision Transformer for Efficient Multi-task Learning with Model-Accelerator Co-design","date":"2022-10-26","arxiv_id":"2210.14793","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/m-3-vit-mixture-of-experts-vision-transformer#ran","syntology_url":"https://syntology.ai/paper/2210.14793","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.14793"}},"official":{"repos":["vita-group/m3vit"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/automoe-neural-architecture-search-for","slug":"automoe-neural-architecture-search-for","title":"AutoMoE: Heterogeneous Mixture-of-Experts with Adaptive Computation for Efficient Neural Machine Translation","date":"2022-10-14","arxiv_id":"2210.07535","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":2,"n_ran_checked":3,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/automoe-neural-architecture-search-for#ran","syntology_url":"https://syntology.ai/paper/2210.07535","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07535"}},"official":{"repos":["microsoft/automoe"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/meta-dmoe-adapting-to-domain-shift-by-meta","slug":"meta-dmoe-adapting-to-domain-shift-by-meta","title":"Meta-DMoE: Adapting to Domain Shift by Meta-Distillation from Mixture-of-Experts","date":"2022-10-08","arxiv_id":"2210.03885","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/meta-dmoe-adapting-to-domain-shift-by-meta#ran","syntology_url":"https://syntology.ai/paper/2210.03885","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.03885"}},"official":{"repos":["n3il666/meta-dmoe"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/admoe-anomaly-detection-with-mixture-of","slug":"admoe-anomaly-detection-with-mixture-of","title":"ADMoE: Anomaly Detection with Mixture-of-Experts from Noisy Labels","date":"2022-08-24","arxiv_id":"2208.11290","repositories_listed":1,"syntology":null},{"url":"/paper/mask-and-reason-pre-training-knowledge-graph","slug":"mask-and-reason-pre-training-knowledge-graph","title":"Mask and Reason: Pre-Training Knowledge Graph Transformers for Complex Logical Queries","date":"2022-08-16","arxiv_id":"2208.07638","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mask-and-reason-pre-training-knowledge-graph#ran","syntology_url":"https://syntology.ai/paper/2208.07638","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.07638"}},"official":{"repos":["thudm/kgtransformer"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-soccer-juggling-skills-with-layer","slug":"learning-soccer-juggling-skills-with-layer","title":"Learning Soccer Juggling Skills with Layer-wise Mixture-of-Experts","date":"2022-07-24","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/rome-role-aware-mixture-of-expert-transformer","slug":"rome-role-aware-mixture-of-expert-transformer","title":"RoME: Role-aware Mixture-of-Expert Transformer for Text-to-Video Retrieval","date":"2022-06-26","arxiv_id":"2206.12845","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-expert-models-for-personalization-in","slug":"adaptive-expert-models-for-personalization-in","title":"Adaptive Expert Models for Personalization in Federated Learning","date":"2022-06-15","arxiv_id":"2206.07832","repositories_listed":1,"syntology":null},{"url":"/paper/uni-perceiver-moe-learning-sparse-generalist","slug":"uni-perceiver-moe-learning-sparse-generalist","title":"Uni-Perceiver-MoE: Learning Sparse Generalist Models with Conditional MoEs","date":"2022-06-09","arxiv_id":"2206.04674","repositories_listed":1,"syntology":null},{"url":"/paper/patcher-patch-transformers-with-mixture-of","slug":"patcher-patch-transformers-with-mixture-of","title":"Patcher: Patch Transformers with Mixture of Experts for Precise Medical Image Segmentation","date":"2022-06-03","arxiv_id":"2206.01741","repositories_listed":1,"syntology":null},{"url":"/paper/eliciting-transferability-in-multi-task","slug":"eliciting-transferability-in-multi-task","title":"Eliciting and Understanding Cross-Task Skills with Task-Level Mixture-of-Experts","date":"2022-05-25","arxiv_id":"2205.12701","repositories_listed":1,"syntology":null},{"url":"/paper/sparse-mixers-combining-moe-and-mixing-to","slug":"sparse-mixers-combining-moe-and-mixing-to","title":"Sparse Mixers: Combining MoE and Mixing to build a more efficient BERT","date":"2022-05-24","arxiv_id":"2205.12399","repositories_listed":1,"syntology":null},{"url":"/paper/se-moe-a-scalable-and-efficient-mixture-of","slug":"se-moe-a-scalable-and-efficient-mixture-of","title":"MoESys: A Distributed and Efficient Mixture-of-Experts Training and Inference System for Internet Services","date":"2022-05-20","arxiv_id":"2205.10034","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/se-moe-a-scalable-and-efficient-mixture-of#ran","syntology_url":"https://syntology.ai/paper/2205.10034","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.10034"}},"official":{"repos":["PaddlePaddle/FleetX"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/addressing-confounding-feature-issue-for","slug":"addressing-confounding-feature-issue-for","title":"Addressing Confounding Feature Issue for Causal Recommendation","date":"2022-05-13","arxiv_id":"2205.06532","repositories_listed":1,"syntology":null},{"url":"/paper/table-based-fact-verification-with-self-1","slug":"table-based-fact-verification-with-self-1","title":"Table-based Fact Verification with Self-adaptive Mixture of Experts","date":"2022-04-19","arxiv_id":"2204.08753","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/table-based-fact-verification-with-self-1#ran","syntology_url":"https://syntology.ai/paper/2204.08753","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.08753"}},"official":{"repos":["thumlp/samoe"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/stablemoe-stable-routing-strategy-for-mixture-1","slug":"stablemoe-stable-routing-strategy-for-mixture-1","title":"StableMoE: Stable Routing Strategy for Mixture of Experts","date":"2022-04-18","arxiv_id":"2204.08396","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":3,"n_ran_checked":2,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/stablemoe-stable-routing-strategy-for-mixture-1#ran","syntology_url":"https://syntology.ai/paper/2204.08396","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.08396"}},"official":{"repos":["hunter-ddm/stablemoe"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/moebert-from-bert-to-mixture-of-experts-via-1","slug":"moebert-from-bert-to-mixture-of-experts-via-1","title":"MoEBERT: from BERT to Mixture-of-Experts via Importance-Guided Adaptation","date":"2022-04-15","arxiv_id":"2204.07675","repositories_listed":1,"syntology":null},{"url":"/paper/3m-multi-loss-multi-path-and-multi-level","slug":"3m-multi-loss-multi-path-and-multi-level","title":"3M: Multi-loss, Multi-path and Multi-level Neural Networks for speech recognition","date":"2022-04-07","arxiv_id":"2204.03178","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/3m-multi-loss-multi-path-and-multi-level#ran","syntology_url":"https://syntology.ai/paper/2204.03178","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.03178"}},"official":{"repos":["tencent-ailab/3m-asr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-adapt-clinical-sequences-with","slug":"learning-to-adapt-clinical-sequences-with","title":"Learning to Adapt Clinical Sequences with Residual Mixture of Experts","date":"2022-04-06","arxiv_id":"2204.02687","repositories_listed":1,"syntology":null},{"url":"/paper/combining-spectral-and-self-supervised","slug":"combining-spectral-and-self-supervised","title":"Combining Spectral and Self-Supervised Features for Low Resource Speech Recognition and Translation","date":"2022-04-05","arxiv_id":"2204.02470","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-and-degradation-adaptive-network","slug":"efficient-and-degradation-adaptive-network","title":"Efficient and Degradation-Adaptive Network for Real-World Image Super-Resolution","date":"2022-03-27","arxiv_id":"2203.14216","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/efficient-and-degradation-adaptive-network#ran","syntology_url":"https://syntology.ai/paper/2203.14216","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.14216"}},"official":{"repos":["csjliang/dasr"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/build-a-robust-qa-system-with-transformer","slug":"build-a-robust-qa-system-with-transformer","title":"Build a Robust QA System with Transformer-based Mixture of Experts","date":"2022-03-20","arxiv_id":"2204.09598","repositories_listed":1,"syntology":null},{"url":"/paper/summareranker-a-multi-task-mixture-of-experts-1","slug":"summareranker-a-multi-task-mixture-of-experts-1","title":"SummaReranker: A Multi-Task Mixture-of-Experts Re-ranking Framework for Abstractive Summarization","date":"2022-03-13","arxiv_id":"2203.06569","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/summareranker-a-multi-task-mixture-of-experts-1#ran","syntology_url":"https://syntology.ai/paper/2203.06569","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.06569"}},"official":{"repos":["ntunlp/summareranker"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mdfend-multi-domain-fake-news-detection","slug":"mdfend-multi-domain-fake-news-detection","title":"MDFEND: Multi-domain Fake News Detection","date":"2022-01-04","arxiv_id":"2201.00987","repositories_listed":1,"syntology":null},{"url":"/paper/meta-mimicking-embedding-via-others","slug":"meta-mimicking-embedding-via-others","title":"Mimic Embedding via Adaptive Aggregation: Learning Generalizable Person Re-identification","date":"2021-12-16","arxiv_id":"2112.08684","repositories_listed":1,"syntology":null},{"url":"/paper/specializing-versatile-skill-libraries-using","slug":"specializing-versatile-skill-libraries-using","title":"Specializing Versatile Skill Libraries using Local Mixture of Experts","date":"2021-12-08","arxiv_id":"2112.04216","repositories_listed":1,"syntology":null},{"url":"/paper/elucidating-noisy-data-via-uncertainty-aware","slug":"elucidating-noisy-data-via-uncertainty-aware","title":"Elucidating Robust Learning with Uncertainty-Aware Corruption Pattern Estimation","date":"2021-11-02","arxiv_id":"2111.01632","repositories_listed":1,"syntology":null},{"url":"/paper/p-adapters-robustly-extracting-factual-1","slug":"p-adapters-robustly-extracting-factual-1","title":"P-Adapters: Robustly Extracting Factual Information from Language Models with Diverse Prompts","date":"2021-10-14","arxiv_id":"2110.07280","repositories_listed":1,"syntology":null},{"url":"/paper/hydrasum-disentangling-stylistic-features-in-1","slug":"hydrasum-disentangling-stylistic-features-in-1","title":"HydraSum: Disentangling Stylistic Features in Text Summarization using Multi-Decoder Models","date":"2021-10-08","arxiv_id":"2110.04400","repositories_listed":1,"syntology":{"n":9,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/hydrasum-disentangling-stylistic-features-in-1#ran","syntology_url":"https://syntology.ai/paper/2110.04400","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.04400"}},"official":{"repos":["salesforce/hydra-sum"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/taming-sparsely-activated-transformer-with","slug":"taming-sparsely-activated-transformer-with","title":"Taming Sparsely Activated Transformer with Stochastic Experts","date":"2021-10-08","arxiv_id":"2110.04260","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/taming-sparsely-activated-transformer-with#ran","syntology_url":"https://syntology.ai/paper/2110.04260","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.04260"}},"official":{"repos":["microsoft/stochastic-mixture-of-experts"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sparse-moes-meet-efficient-ensembles","slug":"sparse-moes-meet-efficient-ensembles","title":"Sparse MoEs meet Efficient Ensembles","date":"2021-10-07","arxiv_id":"2110.03360","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sparse-moes-meet-efficient-ensembles#ran","syntology_url":"https://syntology.ai/paper/2110.03360","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.03360"}},"official":{"repos":["google-research/vmoe"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scalable-and-efficient-moe-training-for","slug":"scalable-and-efficient-moe-training-for","title":"Scalable and Efficient MoE Training for Multitask Multilingual Models","date":"2021-09-22","arxiv_id":"2109.10465","repositories_listed":1,"syntology":null},{"url":"/paper/universal-simultaneous-machine-translation","slug":"universal-simultaneous-machine-translation","title":"Universal Simultaneous Machine Translation with Mixture-of-Experts Wait-k Policy","date":"2021-09-11","arxiv_id":"2109.05238","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/universal-simultaneous-machine-translation#ran","syntology_url":"https://syntology.ai/paper/2109.05238","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.05238"}},"official":{"repos":["ictnlp/moe-waitk"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-mixture-of-experts-model-for-antonym","slug":"a-mixture-of-experts-model-for-antonym","title":"A Mixture-of-Experts Model for Antonym-Synonym Discrimination","date":"2021-08-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/few-shot-and-continual-learning-with","slug":"few-shot-and-continual-learning-with","title":"Few-Shot and Continual Learning with Attentive Independent Mechanisms","date":"2021-07-29","arxiv_id":"2107.14053","repositories_listed":1,"syntology":null},{"url":"/paper/go-wider-instead-of-deeper","slug":"go-wider-instead-of-deeper","title":"Go Wider Instead of Deeper","date":"2021-07-25","arxiv_id":"2107.11817","repositories_listed":1,"syntology":null},{"url":"/paper/lifelong-mixture-of-variational-autoencoders","slug":"lifelong-mixture-of-variational-autoencoders","title":"Lifelong Mixture of Variational Autoencoders","date":"2021-07-09","arxiv_id":"2107.04694","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lifelong-mixture-of-variational-autoencoders#ran","syntology_url":"https://syntology.ai/paper/2107.04694","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.04694"}},"official":{"repos":["dtuzi123/LifelongMixtureVAEs"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptive-3d-descattering-with-a-dynamic","slug":"adaptive-3d-descattering-with-a-dynamic","title":"Adaptive 3D descattering with a dynamic synthesis network","date":"2021-07-01","arxiv_id":"2107.00484","repositories_listed":1,"syntology":null},{"url":"/paper/heterogeneous-multi-task-learning-with-expert","slug":"heterogeneous-multi-task-learning-with-expert","title":"Heterogeneous Multi-task Learning with Expert Diversity","date":"2021-06-20","arxiv_id":"2106.10595","repositories_listed":1,"syntology":null},{"url":"/paper/scaling-vision-with-sparse-mixture-of-experts","slug":"scaling-vision-with-sparse-mixture-of-experts","title":"Scaling Vision with Sparse Mixture of Experts","date":"2021-06-10","arxiv_id":"2106.05974","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scaling-vision-with-sparse-mixture-of-experts#ran","syntology_url":"https://syntology.ai/paper/2106.05974","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.05974"}},"official":{"repos":["google-research/vmoe"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/joint-retrieval-and-generation-training-for","slug":"joint-retrieval-and-generation-training-for","title":"RetGen: A Joint framework for Retrieval and Grounded Text Generation Modeling","date":"2021-05-14","arxiv_id":"2105.06597","repositories_listed":1,"syntology":null},{"url":"/paper/kdexplainer-a-task-oriented-attention-model","slug":"kdexplainer-a-task-oriented-attention-model","title":"KDExplainer: A Task-oriented Attention Model for Explaining Knowledge Distillation","date":"2021-05-10","arxiv_id":"2105.04181","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/kdexplainer-a-task-oriented-attention-model#ran","syntology_url":"https://syntology.ai/paper/2105.04181","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.04181"}},"official":null}},{"url":"/paper/speechmoe-scaling-to-large-acoustic-models","slug":"speechmoe-scaling-to-large-acoustic-models","title":"SpeechMoE: Scaling to Large Acoustic Models with Dynamic Routing Mixture of Experts","date":"2021-05-07","arxiv_id":"2105.03036","repositories_listed":1,"syntology":null},{"url":"/paper/mice-mixture-of-contrastive-experts-for-1","slug":"mice-mixture-of-contrastive-experts-for-1","title":"MiCE: Mixture of Contrastive Experts for Unsupervised Image Clustering","date":"2021-05-05","arxiv_id":"2105.01899","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/mice-mixture-of-contrastive-experts-for-1#ran","syntology_url":"https://syntology.ai/paper/2105.01899","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.01899"}},"official":{"repos":["TsungWeiTsai/MiCE"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/probabilistic-rainfall-estimation-from","slug":"probabilistic-rainfall-estimation-from","title":"Probabilistic Rainfall Estimation from Automotive Lidar","date":"2021-04-23","arxiv_id":"2104.11467","repositories_listed":1,"syntology":null},{"url":"/paper/probabilistic-mixture-of-experts-for-1","slug":"probabilistic-mixture-of-experts-for-1","title":"Probabilistic Mixture-of-Experts for Efficient Deep Reinforcement Learning","date":"2021-04-19","arxiv_id":"2104.09122","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/probabilistic-mixture-of-experts-for-1#ran","syntology_url":"https://syntology.ai/paper/2104.09122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.09122"}},"official":{"repos":["JieRen98/rlkit-pmoe"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-non-asymptotic-penalization-criterion-for","slug":"a-non-asymptotic-penalization-criterion-for","title":"A non-asymptotic approach for model selection via penalization in high-dimensional mixture of experts models","date":"2021-04-06","arxiv_id":"2104.02640","repositories_listed":1,"syntology":null},{"url":"/paper/vdsm-unsupervised-video-disentanglement-with","slug":"vdsm-unsupervised-video-disentanglement-with","title":"VDSM: Unsupervised Video Disentanglement with State-Space Modeling and Deep Mixtures of Experts","date":"2021-03-12","arxiv_id":"2103.07292","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/vdsm-unsupervised-video-disentanglement-with#ran","syntology_url":"https://syntology.ai/paper/2103.07292","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.07292"}},"official":{"repos":["matthewvowels1/DisentanglingSequences"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/real-time-relevant-recommendation-suggestion","slug":"real-time-relevant-recommendation-suggestion","title":"Real-time Relevant Recommendation Suggestion","date":"2021-03-08","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-variational-autoencoders-for-semi-1","slug":"multimodal-variational-autoencoders-for-semi-1","title":"Multimodal Variational Autoencoders for Semi-Supervised Learning: In Defense of Product-of-Experts","date":"2021-01-18","arxiv_id":"2101.07240","repositories_listed":1,"syntology":null},{"url":"/paper/pfl-moe-personalized-federated-learning-based","slug":"pfl-moe-personalized-federated-learning-based","title":"PFL-MoE: Personalized Federated Learning Based on Mixture of Experts","date":"2020-12-31","arxiv_id":"2012.15589","repositories_listed":1,"syntology":null},{"url":"/paper/taxonomy-of-multimodal-self-supervised","slug":"taxonomy-of-multimodal-self-supervised","title":"Self-Supervised Multimodal Domino: in Search of Biomarkers for Alzheimer's Disease","date":"2020-12-25","arxiv_id":"2012.13623","repositories_listed":1,"syntology":null},{"url":"/paper/a-mixture-of-experts-model-for-learning-multi","slug":"a-mixture-of-experts-model-for-learning-multi","title":"A Mixture-of-Experts Model for Learning Multi-Facet Entity Embeddings","date":"2020-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/an-empirical-study-on-model-agnostic","slug":"an-empirical-study-on-model-agnostic","title":"An Empirical Study on Model-agnostic Debiasing Strategies for Robust Natural Language Inference","date":"2020-10-08","arxiv_id":"2010.03777","repositories_listed":1,"syntology":null},{"url":"/paper/federated-learning-using-a-mixture-of-experts","slug":"federated-learning-using-a-mixture-of-experts","title":"Specialized federated learning using a mixture of experts","date":"2020-10-05","arxiv_id":"2010.02056","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/federated-learning-using-a-mixture-of-experts#ran","syntology_url":"https://syntology.ai/paper/2010.02056","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.02056"}},"official":{"repos":["edvinli/federated-learning-mixture"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/restoring-spatially-heterogeneous-distortions","slug":"restoring-spatially-heterogeneous-distortions","title":"Restoring Spatially-Heterogeneous Distortions using Mixture of Experts Network","date":"2020-09-30","arxiv_id":"2009.14563","repositories_listed":1,"syntology":null},{"url":"/paper/transformer-based-multi-source-domain","slug":"transformer-based-multi-source-domain","title":"Transformer Based Multi-Source Domain Adaptation","date":"2020-09-16","arxiv_id":"2009.07806","repositories_listed":1,"syntology":null},{"url":"/paper/anomaly-detection-by-recombining-gated","slug":"anomaly-detection-by-recombining-gated","title":"Anomaly Detection by Recombining Gated Unsupervised Experts","date":"2020-08-31","arxiv_id":"2008.13763","repositories_listed":1,"syntology":null},{"url":"/paper/making-neural-networks-interpretable-with","slug":"making-neural-networks-interpretable-with","title":"Making Neural Networks Interpretable with Attribution: Application to Implicit Signals Prediction","date":"2020-08-26","arxiv_id":"2008.11406","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/making-neural-networks-interpretable-with#ran","syntology_url":"https://syntology.ai/paper/2008.11406","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.11406"}},"official":{"repos":["deezer/interpretable_nn_attribution"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mixture-of-experts-category-hierarchy-soft","slug":"mixture-of-experts-category-hierarchy-soft","title":"Adversarial Mixture Of Experts with Category Hierarchy Soft Constraint","date":"2020-07-24","arxiv_id":"2007.12349","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-model-consensus-to-generate","slug":"exploring-model-consensus-to-generate","title":"Exploring Model Consensus to Generate Translation Paraphrases","date":"2020-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/task-agnostic-online-reinforcement-learning","slug":"task-agnostic-online-reinforcement-learning","title":"Task-Agnostic Online Reinforcement Learning with an Infinite Mixture of Gaussian Processes","date":"2020-06-19","arxiv_id":"2006.11441","repositories_listed":1,"syntology":null},{"url":"/paper/catching-attention-with-automatic-pull-quote","slug":"catching-attention-with-automatic-pull-quote","title":"Catching Attention with Automatic Pull Quote Selection","date":"2020-05-27","arxiv_id":"2005.13263","repositories_listed":1,"syntology":null},{"url":"/paper/learning-charme-models-with-deep-neural","slug":"learning-charme-models-with-deep-neural","title":"Learning CHARME models with neural networks","date":"2020-02-08","arxiv_id":"2002.03237","repositories_listed":1,"syntology":null},{"url":"/paper/self-routing-capsule-networks","slug":"self-routing-capsule-networks","title":"Self-Routing Capsule Networks","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-mixtures-of-generators-for","slug":"hierarchical-mixtures-of-generators-for","title":"Hierarchical Mixtures of Generators for Adversarial Learning","date":"2019-11-05","arxiv_id":"1911.02069","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/hierarchical-mixtures-of-generators-for#ran","syntology_url":"https://syntology.ai/paper/1911.02069","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.02069"}},"official":{"repos":["alper111/hmog"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/extreme-classification-in-log-memory-using","slug":"extreme-classification-in-log-memory-using","title":"Extreme Classification in Log Memory using Count-Min Sketch: A Case Study of Amazon Search with 50M Products","date":"2019-10-28","arxiv_id":"1910.13830","repositories_listed":1,"syntology":null},{"url":"/paper/mixture-of-experts-variational-autoencoder","slug":"mixture-of-experts-variational-autoencoder","title":"Mixture-of-Experts Variational Autoencoder for Clustering and Generating from Similarity-Based Representations on Single Cell Data","date":"2019-10-17","arxiv_id":"1910.07763","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mixture-of-experts-variational-autoencoder#ran","syntology_url":"https://syntology.ai/paper/1910.07763","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.07763"}},"official":{"repos":["andkopf/MoESimVAE"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-a-mixture-of-granularity-specific","slug":"learning-a-mixture-of-granularity-specific","title":"Learning a Mixture of Granularity-Specific Experts for Fine-Grained Categorization","date":"2019-10-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/mixture-content-selection-for-diverse","slug":"mixture-content-selection-for-diverse","title":"Mixture Content Selection for Diverse Sequence Generation","date":"2019-09-04","arxiv_id":"1909.01953","repositories_listed":1,"syntology":null},{"url":"/paper/expert-sample-consensus-applied-to-camera-re","slug":"expert-sample-consensus-applied-to-camera-re","title":"Expert Sample Consensus Applied to Camera Re-Localization","date":"2019-08-07","arxiv_id":"1908.02484","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/expert-sample-consensus-applied-to-camera-re#ran","syntology_url":"https://syntology.ai/paper/1908.02484","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.02484"}},"official":{"repos":["vislearn/esac"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}}],"record_sha256":"7e7139e7f45f04b7b360402f16deba3b54350302f08da1ecfaef2d588cf4baf9","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}