{"url":"/dataset/multinli","name":"MultiNLI","full_name":"Multi-Genre Natural Language Inference","description_markdown":"The **Multi-Genre Natural Language Inference** (**MultiNLI**) dataset has 433K sentence pairs. Its size and mode of collection are modeled closely like [SNLI](/dataset/snli). MultiNLI offers ten distinct genres (Face-to-face, Telephone, 9/11, Travel, Letters, Oxford University Press, Slate, Verbatim, Goverment and Fiction) of written and spoken English data. There are matched dev/test sets which are derived from the same sources as those in the training set, and mismatched sets which do not closely resemble any seen at training time.\r\n\r\nSource: [Semantic Sentence Matching with Densely-connectedRecurrent and Co-attentive Information](https://arxiv.org/abs/1805.11360)","description_withheld":null,"homepage":"https://cims.nyu.edu/~sbowman/multinli/","introduced_date":"2018-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/a-broad-coverage-challenge-corpus-for","title":"A Broad-Coverage Challenge Corpus for Sentence Understanding through Inference","first_author":"Adina Williams","url":null},"license":{"name":"Custom (multiple, see the paper)","url":"https://cims.nyu.edu/~sbowman/multinli/paper.pdf"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Text Generation","url":"/task/text-generation","datasets_with_task":"/datasets/task/text-generation"},{"name":"Natural Language Inference","url":"/task/natural-language-inference","datasets_with_task":"/datasets/task/natural-language-inference"}],"languages":[{"name":"Vietnamese","url":"/datasets/language/vietnamese"}],"variants":["MultiNLI MisMatched dev","MultiNLI Matched dev","MultiNLI Dev MisMatched","MultiNLI Dev Matched","MNLI-mm","mnli_mismatched","MNLI-m","MNLI","MultiNLI-mismatched","MultiNLI-matched","multi_nli","MultiNLI Dev","MultiNLI"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/multi_nli","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/multi_nli_mismatch","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/nyu-mll/multi_nli","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/nyu-mll/multi_nli_mismatch","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/facebookresearch/ParlAI","url":"https://parl.ai/docs/tasks.html#multinli","frameworks":["pytorch"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/multi_nli","frameworks":["tf","jax"]}],"num_papers_in_archive":1830,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/natural-language-inference-on-multinli","task":"Natural Language Inference","dataset_variant":"MultiNLI","rows":67,"metrics":["Matched","Mismatched","Accuracy","Dev Matched","Dev Mismatched"],"first_row_in_archive_order":{"model":"Turing NLR v5 XXL 5.4B (fine-tuned)","paper":null,"metrics":{"Matched":"92.6","Mismatched":"92.4"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/natural-language-inference-on-multinli-dev","task":"Natural Language Inference","dataset_variant":"MultiNLI Dev","rows":10,"metrics":["Matched","Mismatched"],"first_row_in_archive_order":{"model":"TinyBERT-6 67M","paper":"/paper/190910351","metrics":{"Matched":"84.5","Mismatched":"84.5"},"code_links":[{"title":"PaddlePaddle/PaddleNLP","url":"https://github.com/PaddlePaddle/PaddleNLP/tree/develop/paddlenlp/transformers/tinybert"},{"title":"huawei-noah/Pretrained-Language-Model","url":"https://github.com/huawei-noah/Pretrained-Language-Model/tree/master/TinyBERT"},{"title":"mindspore-ai/models","url":"https://github.com/mindspore-ai/models/tree/master/official/nlp/tinybert"},{"title":"mkavim/finetune_bert","url":"https://github.com/mkavim/finetune_bert"},{"title":"MindCode-4/code-5","url":"https://github.com/MindCode-4/code-5/tree/main/tinybert"},{"title":"pwc-1/Paper-9","url":"https://github.com/pwc-1/Paper-9/tree/main/1/tinybert"},{"title":"xiaolilaoli/tiny_bert_ms","url":"https://github.com/xiaolilaoli/tiny_bert_ms"},{"title":"pwc-1/Paper-10","url":"https://github.com/pwc-1/Paper-10/tree/main/tinybert"},{"title":"graison-thomas/TinyFinBERT","url":"https://github.com/graison-thomas/TinyFinBERT"},{"title":"2023-MindSpore-1/ms-code-166","url":"https://github.com/2023-MindSpore-1/ms-code-166"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/natural-language-inference-on-multi-nli","task":"Natural Language Inference","dataset_variant":"multi_nli","rows":0,"metrics":["Accuracy","F1 Macro","F1 Micro","F1 Weighted","Precision Macro","Precision Micro","Precision Weighted","Recall Macro","Recall Micro","Recall Weighted","loss"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-generation-on-mnli","task":"Text Generation","dataset_variant":"MNLI","rows":0,"metrics":["acc"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/first-train-to-generate-then-generate-to","title":"First Train to Generate, then Generate to Train: UnitedSynT5 for Few-Shot NLI","date":"2024-12-12","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/generative-pretrained-structured-transformers","title":"Generative Pretrained Structured Transformers: Unsupervised Syntactic Language Models at Scale","date":"2024-03-13","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":4,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/not-all-layers-are-equally-as-important-every","title":"Not all layers are equally as important: Every Layer Counts BERT","date":"2023-11-03","rows_on_this_dataset":4,"code_links":0,"syntology":null},{"paper":"/paper/lm-cppf-paraphrasing-guided-data-augmentation","title":"LM-CPPF: Paraphrasing-Guided Data Augmentation for Contrastive Prompt-Based Few-Shot Fine-Tuning","date":"2023-05-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/lamini-lm-a-diverse-herd-of-distilled-models","title":"LaMini-LM: A Diverse Herd of Distilled Models from Large-Scale Instructions","date":"2023-04-27","rows_on_this_dataset":5,"code_links":1,"syntology":null},{"paper":"/paper/llm-int8-8-bit-matrix-multiplication-for","title":"LLM.int8(): 8-bit Matrix Multiplication for Transformers at Scale","date":"2022-08-15","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":2,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/adversarial-self-attention-for-language","title":"Adversarial Self-Attention for Language Understanding","date":"2022-06-25","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/prune-once-for-all-sparse-pre-trained","title":"Prune Once for All: Sparse Pre-Trained Language Models","date":"2021-11-10","rows_on_this_dataset":9,"code_links":2,"syntology":null},{"paper":"/paper/charformer-fast-character-transformers-via","title":"Charformer: Fast Character Transformers via Gradient-based Subword Tokenization","date":"2021-06-23","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":7,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/pay-attention-to-mlps","title":"Pay Attention to MLPs","date":"2021-05-17","rows_on_this_dataset":1,"code_links":20,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":44,"samples_ran":34,"samples_unverified":10,"pointer_only_for_licence":11,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/fnet-mixing-tokens-with-fourier-transforms","title":"FNet: Mixing Tokens with Fourier Transforms","date":"2021-05-09","rows_on_this_dataset":2,"code_links":12,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/how-to-train-bert-with-an-academic-budget","title":"How to Train BERT with an Academic Budget","date":"2021-04-15","rows_on_this_dataset":1,"code_links":4,"syntology":null},{"paper":"/paper/informer-transformer-likes-informed-attention","title":"RealFormer: Transformer Likes Residual Attention","date":"2020-12-21","rows_on_this_dataset":1,"code_links":5,"syntology":null},{"paper":"/paper/a-statistical-framework-for-low-bitwidth","title":"A Statistical Framework for Low-bitwidth Training of Deep Neural Networks","date":"2020-10-27","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":1,"samples_unverified":3,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/big-bird-transformers-for-longer-sequences","title":"Big Bird: Transformers for Longer Sequences","date":"2020-07-28","rows_on_this_dataset":1,"code_links":14,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":10,"samples_unverified":5,"pointer_only_for_licence":11,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/squeezebert-what-can-computer-vision-teach","title":"SqueezeBERT: What can computer vision teach NLP about efficient neural networks?","date":"2020-06-19","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/deberta-decoding-enhanced-bert-with","title":"DeBERTa: Decoding-enhanced BERT with Disentangled Attention","date":"2020-06-05","rows_on_this_dataset":1,"code_links":14,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":4,"samples_unverified":9,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/what-do-questions-exactly-ask-mfae-duplicate","title":"What Do Questions Exactly Ask? MFAE: Duplicate Question Identification with Multi-Fusion Asking Emphasis","date":"2020-05-07","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/smart-robust-and-efficient-fine-tuning-for","title":"SMART: Robust and Efficient Fine-Tuning for Pre-trained Natural Language Models through Principled Regularized Optimization","date":"2019-11-08","rows_on_this_dataset":6,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":6,"samples_unverified":2,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/exploring-the-limits-of-transfer-learning","title":"Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer","date":"2019-10-23","rows_on_this_dataset":7,"code_links":57,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":31,"samples_ran":2,"samples_unverified":29,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/q8bert-quantized-8bit-bert","title":"Q8BERT: Quantized 8Bit BERT","date":"2019-10-14","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":3,"samples_unverified":8,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/albert-a-lite-bert-for-self-supervised","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","date":"2019-09-26","rows_on_this_dataset":1,"code_links":48,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":126,"samples_ran":46,"samples_unverified":80,"pointer_only_for_licence":22,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/190910351","title":"TinyBERT: Distilling BERT for Natural Language Understanding","date":"2019-09-23","rows_on_this_dataset":3,"code_links":10,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/q-bert-hessian-based-ultra-low-precision","title":"Q-BERT: Hessian Based Ultra Low Precision Quantization of BERT","date":"2019-09-12","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/structbert-incorporating-language-structures","title":"StructBERT: Incorporating Language Structures into Pre-training for Deep Language Understanding","date":"2019-08-13","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/ernie-20-a-continual-pre-training-framework","title":"ERNIE 2.0: A Continual Pre-training Framework for Language Understanding","date":"2019-07-29","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/roberta-a-robustly-optimized-bert-pretraining","title":"RoBERTa: A Robustly Optimized BERT Pretraining Approach","date":"2019-07-26","rows_on_this_dataset":2,"code_links":67,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":48,"samples_ran":22,"samples_unverified":26,"pointer_only_for_licence":23,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/spanbert-improving-pre-training-by","title":"SpanBERT: Improving Pre-training by Representing and Predicting Spans","date":"2019-07-24","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":3,"samples_unverified":12,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/xlnet-generalized-autoregressive-pretraining","title":"XLNet: Generalized Autoregressive Pretraining for Language Understanding","date":"2019-06-19","rows_on_this_dataset":1,"code_links":27,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":24,"samples_ran":10,"samples_unverified":14,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ernie-enhanced-language-representation-with","title":"ERNIE: Enhanced Language Representation with Informative Entities","date":"2019-05-17","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/improving-multi-task-deep-neural-networks-via","title":"Improving Multi-Task Deep Neural Networks via Knowledge Distillation for Natural Language Understanding","date":"2019-04-20","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/multi-task-deep-neural-networks-for-natural","title":"Multi-Task Deep Neural Networks for Natural Language Understanding","date":"2019-01-31","rows_on_this_dataset":1,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":5,"samples_unverified":8,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/attention-boosted-sequential-inference-model","title":"Attention Boosted Sequential Inference Model","date":"2018-12-05","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/combining-similarity-features-and-deep","title":"Combining Similarity Features and Deep Representation Learning for Stance Detection in the Context of Checking Fake News","date":"2018-11-02","rows_on_this_dataset":3,"code_links":3,"syntology":null},{"paper":"/paper/bert-pre-training-of-deep-bidirectional","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","date":"2018-10-11","rows_on_this_dataset":1,"code_links":534,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":659,"samples_ran":204,"samples_unverified":455,"pointer_only_for_licence":149,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/training-complex-models-with-multi-task-weak","title":"Training Complex Models with Multi-Task Weak Supervision","date":"2018-10-05","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/improving-language-understanding-by","title":"Improving Language Understanding by Generative Pre-Training","date":"2018-06-11","rows_on_this_dataset":1,"code_links":13,"syntology":null},{"paper":"/paper/baseline-needs-more-love-on-simple-word","title":"Baseline Needs More Love: On Simple Word-Embedding-Based Models and Associated Pooling Mechanisms","date":"2018-05-24","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/glue-a-multi-task-benchmark-and-analysis","title":"GLUE: A Multi-Task Benchmark and Analysis Platform for Natural Language Understanding","date":"2018-04-20","rows_on_this_dataset":1,"code_links":11,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":5,"samples_unverified":9,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-general-purpose-distributed-sentence","title":"Learning General Purpose Distributed Sentence Representations via Large Scale Multi-task Learning","date":"2018-03-30","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":23,"samples_harvested":1061,"samples_ran":376,"samples_unverified":685,"pointer_only_for_licence":244,"papers_with_no_sample_that_ran":3,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}