{"url":"/dataset/anli","name":"ANLI","full_name":"Adversarial NLI","description_markdown":"The Adversarial Natural Language Inference (**ANLI**, Nie et al.) is a new large-scale NLI benchmark dataset, collected via an iterative, adversarial human-and-model-in-the-loop procedure. Particular, the data is selected to be difficult to the state-of-the-art models, including BERT and RoBERTa.\r\n\r\nSource: [The Microsoft Toolkit of Multi-Task Deep Neural Networks for Natural Language Understanding](https://arxiv.org/abs/2002.07972)\r\nImage Source: [https://arxiv.org/pdf/1910.14599.pdf](https://arxiv.org/pdf/1910.14599.pdf)","description_withheld":null,"homepage":"https://github.com/facebookresearch/anli","introduced_date":"2019-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/adversarial-nli-a-new-benchmark-for-natural","title":"Adversarial NLI: A New Benchmark for Natural Language Understanding","first_author":"Yixin Nie","url":null},"license":{"name":"CC BY-NC 4.0","url":"https://github.com/facebookresearch/anli/blob/master/LICENSE"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Natural Language Inference","url":"/task/natural-language-inference","datasets_with_task":"/datasets/task/natural-language-inference"}],"languages":[],"variants":["ANLI test","ANLI","ANLI-all","ANLI-r3"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/facebook/anli","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/anli","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/KETI-AIR/anli","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/KETI-AIR/kor_anli","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/facebookresearch/ParlAI","url":"https://parl.ai/docs/tasks.html#adversarial-natural-language-inference-anli-corpus","frameworks":["pytorch"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/anli","frameworks":["tf","jax"]},{"repo":"https://github.com/facebookresearch/anli","url":"https://github.com/facebookresearch/anli","frameworks":["pytorch"]}],"num_papers_in_archive":287,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/natural-language-inference-on-anli-test","task":"Natural Language Inference","dataset_variant":"ANLI test","rows":25,"metrics":["A1","A2","A3"],"first_row_in_archive_order":{"model":"T5-3B (explanation prompting)","paper":"/paper/prompting-for-explanations-improves","metrics":{"A1":"81.8","A2":"72.5","A3":"74.8"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/natural-language-inference-on-anli","task":"Natural Language Inference","dataset_variant":"ANLI","rows":0,"metrics":["Accuracy","F1 Macro","F1 Micro","F1 Weighted","Precision Macro","Precision Micro","Precision Weighted","Recall Macro","Recall Micro","Recall Weighted","loss"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/natural-language-inference-on-anli-r3","task":"Natural Language Inference","dataset_variant":"ANLI-r3","rows":0,"metrics":["Accuracy"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/a-systematic-study-and-comprehensive","title":"A Systematic Study and Comprehensive Evaluation of ChatGPT on Benchmark Datasets","date":"2023-05-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/the-cot-collection-improving-zero-shot-and","title":"The CoT Collection: Improving Zero-shot and Few-shot Learning of Language Models via Chain-of-Thought Fine-Tuning","date":"2023-05-23","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/palm-2-technical-report-1","title":"PaLM 2 Technical Report","date":"2023-05-17","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/prompting-for-explanations-improves","title":"Prompting for explanations improves Adversarial NLI. Is this true? {Yes} it is {true} because {it weakens superficial cues}","date":"2023-05-01","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/bloomberggpt-a-large-language-model-for","title":"BloombergGPT: A Large Language Model for Finance","date":"2023-03-30","rows_on_this_dataset":4,"code_links":2,"syntology":null},{"paper":"/paper/exploring-the-benefits-of-training-expert","title":"Exploring the Benefits of Training Expert Language Models over Instruction Tuning","date":"2023-02-07","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/knowledge-in-context-towards-knowledgeable","title":"Knowledge-in-Context: Towards Knowledgeable Semi-Parametric Language Models","date":"2022-10-28","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/large-language-models-can-self-improve","title":"Large Language Models Can Self-Improve","date":"2022-10-20","rows_on_this_dataset":6,"code_links":0,"syntology":null},{"paper":"/paper/guess-the-instruction-making-language-models","title":"Guess the Instruction! Flipped Learning Makes Language Models Stronger Zero-Shot Learners","date":"2022-10-06","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/infobert-improving-robustness-of-language-1","title":"InfoBERT: Improving Robustness of Language Models from An Information Theoretic Perspective","date":"2020-10-05","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":4,"samples_unverified":0,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/language-models-are-few-shot-learners","title":"Language Models are Few-Shot Learners","date":"2020-05-28","rows_on_this_dataset":1,"code_links":67,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":65,"samples_ran":15,"samples_unverified":50,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/adversarial-training-for-large-neural","title":"Adversarial Training for Large Neural Language Models","date":"2020-04-20","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/roberta-a-robustly-optimized-bert-pretraining","title":"RoBERTa: A Robustly Optimized BERT Pretraining Approach","date":"2019-07-26","rows_on_this_dataset":1,"code_links":67,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":48,"samples_ran":22,"samples_unverified":26,"pointer_only_for_licence":23,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/xlnet-generalized-autoregressive-pretraining","title":"XLNet: Generalized Autoregressive Pretraining for Language Understanding","date":"2019-06-19","rows_on_this_dataset":1,"code_links":27,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":24,"samples_ran":10,"samples_unverified":14,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":5,"samples_harvested":144,"samples_ran":54,"samples_unverified":90,"pointer_only_for_licence":37,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}