{"url":"/task/masked-language-modeling","name":"Masked Language Modeling","slug":"masked-language-modeling","description_markdown":null,"categories":[],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":475,"papers_with_code":254,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":7,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[{"url":"/dataset/difair","name":"DiFair","full_name":"","num_papers_in_archive":3},{"url":"/dataset/latamxix","name":"LatamXIX","full_name":"19th Century Latin American Spanish Newspaper Corpus with LLM OCR Correction","num_papers_in_archive":2},{"url":"/dataset/blbooks","name":"blbooks","full_name":"The British Library Books","num_papers_in_archive":1},{"url":"/dataset/geneutral","name":"GENEUTRAL","full_name":"","num_papers_in_archive":1},{"url":"/dataset/genter","name":"GENTER","full_name":"GEnder Name TEmplates with pRonouns","num_papers_in_archive":1},{"url":"/dataset/gentypes","name":"GENTYPES","full_name":"Gender Stereotypes","num_papers_in_archive":1},{"url":"/dataset/spanish-corpus-xix","name":"Spanish Corpus XIX","full_name":"19th Century Spanish Corpus","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":254,"tagged_in_all":475,"items":[{"url":"/paper/electra-pre-training-text-encoders-as-1","title":"ELECTRA: Pre-training Text Encoders as Discriminators Rather Than Generators","date":"2020-03-23","arxiv_id":"2003.10555","repositories_listed":19,"syntology":{"n":40,"n_ran":26,"n_unverified":14,"n_pointer_only":10}},{"url":"/paper/lxmert-learning-cross-modality-encoder","title":"LXMERT: Learning Cross-Modality Encoder Representations from Transformers","date":"2019-08-20","arxiv_id":"1908.07490","repositories_listed":9,"syntology":{"n":15,"n_ran":4,"n_unverified":11,"n_pointer_only":3}},{"url":"/paper/mpnet-masked-and-permuted-pre-training-for","title":"MPNet: Masked and Permuted Pre-training for Language Understanding","date":"2020-04-20","arxiv_id":"2004.09297","repositories_listed":7,"syntology":{"n":8,"n_ran":5,"n_unverified":3,"n_pointer_only":6}},{"url":"/paper/on-the-cross-lingual-transferability-of","title":"On the Cross-lingual Transferability of Monolingual Representations","date":"2019-10-25","arxiv_id":"1910.11856","repositories_listed":7,"syntology":{"n":7,"n_ran":0,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/uniter-learning-universal-image-text-1","title":"UNITER: UNiversal Image-TExt Representation Learning","date":"2019-09-25","arxiv_id":"1909.11740","repositories_listed":7,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/language-agnostic-bert-sentence-embedding","title":"Language-agnostic BERT Sentence Embedding","date":"2020-07-03","arxiv_id":"2007.01852","repositories_listed":6,"syntology":null},{"url":"/paper/realm-retrieval-augmented-language-model-pre","title":"REALM: Retrieval-Augmented Language Model Pre-Training","date":"2020-02-10","arxiv_id":"2002.08909","repositories_listed":6,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/informer-transformer-likes-informed-attention","title":"RealFormer: Transformer Likes Residual Attention","date":"2020-12-21","arxiv_id":"2012.11747","repositories_listed":5,"syntology":null},{"url":"/paper/layoutlmv3-pre-training-for-document-ai-with","title":"LayoutLMv3: Pre-training for Document AI with Unified Text and Image Masking","date":"2022-04-18","arxiv_id":"2204.08387","repositories_listed":4,"syntology":null},{"url":"/paper/w2v-bert-combining-contrastive-learning-and","title":"W2v-BERT: Combining Contrastive Learning and Masked Language Modeling for Self-Supervised Speech Pre-Training","date":"2021-08-07","arxiv_id":"2108.06209","repositories_listed":4,"syntology":null},{"url":"/paper/talking-heads-attention","title":"Talking-Heads Attention","date":"2020-03-05","arxiv_id":"2003.02436","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/zero-shot-video-question-answering-via-frozen","title":"Zero-Shot Video Question Answering via Frozen Bidirectional Language Models","date":"2022-06-16","arxiv_id":"2206.08155","repositories_listed":3,"syntology":{"n":34,"n_ran":14,"n_unverified":20,"n_pointer_only":1}},{"url":"/paper/hero-hierarchical-encoder-for-video-language","title":"HERO: Hierarchical Encoder for Video+Language Omni-representation Pre-training","date":"2020-05-01","arxiv_id":"2005.00200","repositories_listed":3,"syntology":{"n":13,"n_ran":5,"n_unverified":8,"n_pointer_only":8}},{"url":"/paper/self-supervised-log-parsing","title":"Self-Supervised Log Parsing","date":"2020-03-17","arxiv_id":"2003.07905","repositories_listed":3,"syntology":null},{"url":"/paper/simple-and-effective-masked-diffusion","title":"Simple and Effective Masked Diffusion Language Models","date":"2024-06-11","arxiv_id":"2406.07524","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/syllable-discovery-and-cross-lingual","title":"Syllable Discovery and Cross-Lingual Generalization in a Visually Grounded, Self-Supervised Speech Model","date":"2023-05-19","arxiv_id":"2305.11435","repositories_listed":2,"syntology":{"n":16,"n_ran":6,"n_unverified":10,"n_pointer_only":0}},{"url":"/paper/self-supervised-vision-language-pretraining","title":"Self-supervised vision-language pretraining for Medical visual question answering","date":"2022-11-24","arxiv_id":"2211.13594","repositories_listed":2,"syntology":{"n":10,"n_ran":6,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/deep-bidirectional-language-knowledge-graph","title":"Deep Bidirectional Language-Knowledge Graph Pretraining","date":"2022-10-17","arxiv_id":"2210.09338","repositories_listed":2,"syntology":{"n":18,"n_ran":2,"n_unverified":16,"n_pointer_only":0}},{"url":"/paper/mixture-of-attention-heads-selecting","title":"Mixture of Attention Heads: Selecting Attention Heads Per Token","date":"2022-10-11","arxiv_id":"2210.05144","repositories_listed":2,"syntology":{"n":17,"n_ran":6,"n_unverified":11,"n_pointer_only":0}},{"url":"/paper/image-as-a-foreign-language-beit-pretraining","title":"Image as a Foreign Language: BEiT Pretraining for All Vision and Vision-Language Tasks","date":"2022-08-22","arxiv_id":"2208.10442","repositories_listed":2,"syntology":null},{"url":"/paper/unsupervised-pre-training-of-graph","title":"Unsupervised pre-training of graph transformers on patient population graphs","date":"2022-07-21","arxiv_id":"2207.10603","repositories_listed":2,"syntology":null},{"url":"/paper/smt-dta-improving-drug-target-affinity","title":"SSM-DTA: Breaking the Barriers of Data Scarcity in Drug-Target Affinity Prediction","date":"2022-06-20","arxiv_id":"2206.09818","repositories_listed":2,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/an-empirical-study-of-self-supervised","title":"An Empirical Study Of Self-supervised Learning Approaches For Object Detection With Transformers","date":"2022-05-11","arxiv_id":"2205.05543","repositories_listed":2,"syntology":null},{"url":"/paper/vision-language-pre-training-for-boosting","title":"Vision-Language Pre-Training for Boosting Scene Text Detectors","date":"2022-04-29","arxiv_id":"2204.13867","repositories_listed":2,"syntology":{"n":7,"n_ran":5,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/transformer-quality-in-linear-time","title":"Transformer Quality in Linear Time","date":"2022-02-21","arxiv_id":"2202.10447","repositories_listed":2,"syntology":null},{"url":"/paper/dense-to-sparse-gate-for-mixture-of-experts-1","title":"EvoMoE: An Evolutional Mixture-of-Experts Training Framework via Dense-To-Sparse Gate","date":"2021-12-29","arxiv_id":"2112.14397","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/ibot-image-bert-pre-training-with-online","title":"iBOT: Image BERT Pre-Training with Online Tokenizer","date":"2021-11-15","arxiv_id":"2111.07832","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/composable-sparse-fine-tuning-for-cross","title":"Composable Sparse Fine-Tuning for Cross-Lingual Transfer","date":"2021-10-14","arxiv_id":"2110.07560","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":2}},{"url":"/paper/knowledgeable-prompt-tuning-incorporating","title":"Knowledgeable Prompt-tuning: Incorporating Knowledge into Prompt Verbalizer for Text Classification","date":"2021-08-04","arxiv_id":"2108.02035","repositories_listed":2,"syntology":null},{"url":"/paper/luna-linear-unified-nested-attention","title":"Luna: Linear Unified Nested Attention","date":"2021-06-03","arxiv_id":"2106.01540","repositories_listed":2,"syntology":{"n":10,"n_ran":5,"n_unverified":5,"n_pointer_only":5}}],"syntology_records":20,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}