{"url":"/dataset/mednli","name":"MedNLI","full_name":"Medical Natural Language Inference","description_markdown":"The **MedNLI** dataset consists of the sentence pairs developed by Physicians from the Past Medical History section of MIMIC-III clinical notes annotated for Definitely True, Maybe True and Definitely False. The dataset contains 11,232 training, 1,395 development and 1,422 test instances. This provides a natural language inference task (NLI) grounded in the medical history of patients.\n\nSource: [MT-Clinical BERT: Scaling Clinical Information Extraction with Multitask Learning](https://arxiv.org/abs/2004.10220)\nImage Source: [https://arxiv.org/abs/1904.02181](https://arxiv.org/abs/1904.02181)","description_withheld":null,"homepage":"https://physionet.org/content/mednli/1.0.0/","introduced_date":null,"introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"},{"name":"Medical","url":"/datasets/modality/medical"}],"tasks":[{"name":"Few-Shot Learning","url":"/task/few-shot-learning","datasets_with_task":"/datasets/task/few-shot-learning"},{"name":"Natural Language Inference","url":"/task/natural-language-inference","datasets_with_task":"/datasets/task/natural-language-inference"}],"languages":[],"variants":["MedNLI"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/OUTCOMESAI/mednli","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/bigbio/mednli","frameworks":["tf","pytorch","jax"]}],"num_papers_in_archive":9,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/natural-language-inference-on-mednli","task":"Natural Language Inference","dataset_variant":"MedNLI","rows":7,"metrics":["Accuracy","Params (M)"],"first_row_in_archive_order":{"model":"ClinicalMosaic","paper":"/paper/patient-trajectory-prediction-integrating","metrics":{"Accuracy":"86.59","Params (M)":"137"},"code_links":[{"title":"MostHumble/PatientTrajectoryForecasting","url":"https://github.com/MostHumble/PatientTrajectoryForecasting"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/few-shot-learning-on-mednli","task":"Few-Shot Learning","dataset_variant":"MedNLI","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"CoT-T5-11B (1024 Shot)","paper":"/paper/the-cot-collection-improving-zero-shot-and","metrics":{"Accuracy":"78.02"},"code_links":[{"title":"kaistai/cot-collection","url":"https://github.com/kaistai/cot-collection"},{"title":"kaist-lklab/cot-collection","url":"https://github.com/kaist-lklab/cot-collection"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/patient-trajectory-prediction-integrating","title":"Patient Trajectory Prediction: Integrating Clinical Notes with Transformers","date":"2025-02-25","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/biomedgpt-a-unified-and-generalist-biomedical","title":"BiomedGPT: A Generalist Vision-Language Foundation Model for Diverse Biomedical Tasks","date":"2023-05-26","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":1,"samples_unverified":7,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/the-cot-collection-improving-zero-shot-and","title":"The CoT Collection: Improving Zero-shot and Few-shot Learning of Language Models via Chain-of-Thought Fine-Tuning","date":"2023-05-23","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/bioelectra-pretrained-biomedical-text-encoder","title":"BioELECTRA:Pretrained Biomedical text Encoder using Discriminators","date":"2021-06-11","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/scifive-a-text-to-text-transformer-model-for","title":"SciFive: a text-to-text transformer model for biomedical literature","date":"2021-05-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":0,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/characterbert-reconciling-elmo-and-bert-for","title":"CharacterBERT: Reconciling ELMo and BERT for Word-Level Open-Vocabulary Representations From Characters","date":"2020-10-20","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":5,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/saama-research-at-mediqa-2019-pre-trained","title":"Saama Research at MEDIQA 2019: Pre-trained BioBERT with Attention Visualisation for Medical Natural Language Inference","date":"2019-08-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/transfer-learning-in-biomedical-natural","title":"Transfer Learning in Biomedical Natural Language Processing: An Evaluation of BERT and ELMo on Ten Benchmarking Datasets","date":"2019-06-13","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":4,"samples_harvested":23,"samples_ran":6,"samples_unverified":17,"pointer_only_for_licence":7,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}