{"url":"/task/xlm-r","name":"XLM-R","slug":"xlm-r","description_markdown":"XLM-R","categories":[{"name":"Medical","url":"/area/medical"},{"name":"Miscellaneous","url":"/area/miscellaneous"},{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":221,"papers_with_code":99,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":1,"subtasks":0,"parent_tasks":1},"benchmarks":[],"datasets":[{"url":"/dataset/belebele","name":"Belebele","full_name":"","num_papers_in_archive":68}],"subtasks":[],"parent_tasks":[{"url":"/task/language-modelling","name":"Language Modelling"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":99,"tagged_in_all":221,"items":[{"url":"/paper/unsupervised-cross-lingual-representation-1","title":"Unsupervised Cross-lingual Representation Learning at Scale","date":"2019-11-05","arxiv_id":"1911.02116","repositories_listed":35,"syntology":{"n":59,"n_ran":25,"n_unverified":34,"n_pointer_only":52}},{"url":"/paper/adapterhub-a-framework-for-adapting","title":"AdapterHub: A Framework for Adapting Transformers","date":"2020-07-15","arxiv_id":"2007.07779","repositories_listed":9,"syntology":{"n":15,"n_ran":5,"n_unverified":10,"n_pointer_only":12}},{"url":"/paper/massive-a-1m-example-multilingual-natural","title":"MASSIVE: A 1M-Example Multilingual Natural Language Understanding Dataset with 51 Typologically-Diverse Languages","date":"2022-04-18","arxiv_id":"2204.08582","repositories_listed":6,"syntology":{"n":10,"n_ran":1,"n_unverified":9,"n_pointer_only":1}},{"url":"/paper/emotion-classification-in-a-resource","title":"Emotion Classification in a Resource Constrained Language Using Transformer-based Approach","date":"2021-04-17","arxiv_id":"2104.08613","repositories_listed":4,"syntology":null},{"url":"/paper/xlm-v-overcoming-the-vocabulary-bottleneck-in","title":"XLM-V: Overcoming the Vocabulary Bottleneck in Multilingual Masked Language Models","date":"2023-01-25","arxiv_id":"2301.10472","repositories_listed":3,"syntology":null},{"url":"/paper/debertav3-improving-deberta-using-electra","title":"DeBERTaV3: Improving DeBERTa using ELECTRA-Style Pre-Training with Gradient-Disentangled Embedding Sharing","date":"2021-11-18","arxiv_id":"2111.09543","repositories_listed":3,"syntology":{"n":7,"n_ran":0,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/bertweet-a-pre-trained-language-model-for","title":"BERTweet: A pre-trained language model for English Tweets","date":"2020-05-20","arxiv_id":"2005.10200","repositories_listed":3,"syntology":null},{"url":"/paper/mad-x-an-adapter-based-framework-for-multi","title":"MAD-X: An Adapter-Based Framework for Multi-Task Cross-Lingual Transfer","date":"2020-04-30","arxiv_id":"2005.00052","repositories_listed":3,"syntology":null},{"url":"/paper/from-n-grams-to-pre-trained-multilingual","title":"From N-grams to Pre-trained Multilingual Models For Language Identification","date":"2024-10-11","arxiv_id":"2410.08728","repositories_listed":2,"syntology":null},{"url":"/paper/zero-shot-tokenizer-transfer","title":"Zero-Shot Tokenizer Transfer","date":"2024-05-13","arxiv_id":"2405.07883","repositories_listed":2,"syntology":{"n":12,"n_ran":7,"n_unverified":5,"n_pointer_only":4}},{"url":"/paper/focus-effective-embedding-initialization-for","title":"FOCUS: Effective Embedding Initialization for Monolingual Specialization of Multilingual Models","date":"2023-05-23","arxiv_id":"2305.14481","repositories_listed":2,"syntology":null},{"url":"/paper/dumb-a-benchmark-for-smart-evaluation-of","title":"DUMB: A Benchmark for Smart Evaluation of Dutch Models","date":"2023-05-22","arxiv_id":"2305.13026","repositories_listed":2,"syntology":null},{"url":"/paper/greekbart-the-first-pretrained-greek-sequence","title":"GreekBART: The First Pretrained Greek Sequence-to-Sequence Model","date":"2023-04-03","arxiv_id":"2304.00869","repositories_listed":2,"syntology":null},{"url":"/paper/x-2-vlm-all-in-one-pre-trained-model-for","title":"X$^2$-VLM: All-In-One Pre-trained Model For Vision-Language Tasks","date":"2022-11-22","arxiv_id":"2211.12402","repositories_listed":2,"syntology":{"n":6,"n_ran":2,"n_unverified":4,"n_pointer_only":6}},{"url":"/paper/altclip-altering-the-language-encoder-in-clip","title":"AltCLIP: Altering the Language Encoder in CLIP for Extended Language Capabilities","date":"2022-11-12","arxiv_id":"2211.06679","repositories_listed":2,"syntology":{"n":11,"n_ran":2,"n_unverified":9,"n_pointer_only":0}},{"url":"/paper/xeroalign-zero-shot-cross-lingual-transformer","title":"XeroAlign: Zero-Shot Cross-lingual Transformer Alignment","date":"2021-05-06","arxiv_id":"2105.02472","repositories_listed":2,"syntology":null},{"url":"/paper/minilmv2-multi-head-self-attention-relation","title":"MiniLMv2: Multi-Head Self-Attention Relation Distillation for Compressing Pretrained Transformers","date":"2020-12-31","arxiv_id":"2012.15828","repositories_listed":2,"syntology":null},{"url":"/paper/arbert-marbert-deep-bidirectional","title":"ARBERT & MARBERT: Deep Bidirectional Transformers for Arabic","date":"2020-12-27","arxiv_id":"2101.01785","repositories_listed":2,"syntology":null},{"url":"/paper/graph-based-universal-dependency-parsing-in","title":"Applying Occam's Razor to Transformer-Based Dependency Parsing: What Works, What Doesn't, and What is Really Necessary","date":"2020-10-23","arxiv_id":"2010.12699","repositories_listed":2,"syntology":null},{"url":"/paper/bayesian-multilingual-topic-model-for-zero","title":"A Bayesian Multilingual Document Model for Zero-shot Topic Identification and Discovery","date":"2020-07-02","arxiv_id":"2007.01359","repositories_listed":2,"syntology":null},{"url":"/paper/xglue-a-new-benchmark-dataset-for-cross","title":"XGLUE: A New Benchmark Dataset for Cross-lingual Pre-training, Understanding and Generation","date":"2020-04-03","arxiv_id":"2004.01401","repositories_listed":2,"syntology":null},{"url":"/paper/multilingual-encoder-knows-more-than-you","title":"Multilingual Encoder Knows more than You Realize: Shared Weights Pretraining for Extremely Low-Resource Languages","date":"2025-02-15","arxiv_id":"2502.10852","repositories_listed":1,"syntology":null},{"url":"/paper/langsamp-language-script-aware-multilingual","title":"LangSAMP: Language-Script Aware Multilingual Pretraining","date":"2024-09-26","arxiv_id":"2409.18199","repositories_listed":1,"syntology":null},{"url":"/paper/lowrem-a-repository-of-word-embeddings-for-87","title":"GrEmLIn: A Repository of Green Baseline Embeddings for 87 Low-Resource Languages Injected with Multilingual Graph Knowledge","date":"2024-09-26","arxiv_id":"2409.18193","repositories_listed":1,"syntology":null},{"url":"/paper/medical-spoken-named-entity-recognition","title":"Medical Spoken Named Entity Recognition","date":"2024-06-19","arxiv_id":"2406.13337","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-alignment-in-shared-cross-lingual","title":"Exploring Alignment in Shared Cross-lingual Spaces","date":"2024-05-23","arxiv_id":"2405.14535","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/cross-lingual-transfer-robustness-to-lower","title":"Cross-Lingual Transfer Robustness to Lower-Resource Languages on Adversarial Datasets","date":"2024-03-29","arxiv_id":"2403.20056","repositories_listed":1,"syntology":null},{"url":"/paper/cicle-conformal-in-context-learning-for","title":"CICLe: Conformal In-Context Learning for Largescale Multi-Class Food Risk Classification","date":"2024-03-18","arxiv_id":"2403.11904","repositories_listed":1,"syntology":null},{"url":"/paper/kbioxlm-a-knowledge-anchored-biomedical","title":"KBioXLM: A Knowledge-anchored Biomedical Multilingual Pretrained Language Model","date":"2023-11-20","arxiv_id":"2311.11564","repositories_listed":1,"syntology":null},{"url":"/paper/mela-multilingual-evaluation-of-linguistic","title":"MELA: Multilingual Evaluation of Linguistic Acceptability","date":"2023-11-15","arxiv_id":"2311.09033","repositories_listed":1,"syntology":null}],"syntology_records":8,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}