{"url":"/dataset/scitail","name":"SciTail","full_name":null,"description_markdown":"The **SciTail** dataset is an entailment dataset created from multiple-choice science exams and web sentences. Each question and the correct answer choice are converted into an assertive statement to form the hypothesis. We use information retrieval to obtain relevant text from a large text corpus of web sentences, and use these sentences as a premise P. We crowdsource the annotation of such premise-hypothesis pair as supports (entails) or not (neutral), in order to create the SciTail dataset. The dataset contains 27,026 examples with 10,101 examples with entails label and 16,925 examples with neutral label.\r\n\r\nSource: [Allen Institute for AI](https://allenai.org/data/scitail)\r\nImage source: [Allen Institute for AI](https://allenai.org/data/scitail)","description_withheld":null,"homepage":"https://allenai.org/data/scitail","introduced_date":null,"introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Natural Language Inference","url":"/task/natural-language-inference","datasets_with_task":"/datasets/task/natural-language-inference"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["SciTail"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/scitail","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/bigbio/scitail","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/allenai/scitail","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/sci_tail","frameworks":["tf","pytorch","jax"]}],"num_papers_in_archive":9,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/natural-language-inference-on-scitail","task":"Natural Language Inference","dataset_variant":"SciTail","rows":13,"metrics":["Accuracy","Dev Accuracy","% Dev Accuracy","% Test Accuracy"],"first_row_in_archive_order":{"model":"CA-MTL","paper":"/paper/conditionally-adaptive-multi-task-learning","metrics":{"Accuracy":"96.8"},"code_links":[{"title":"CAMTL/CA-MTL","url":"https://github.com/CAMTL/CA-MTL"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/splitee-early-exit-in-deep-neural-networks","title":"SplitEE: Early Exit in Deep Neural Networks with Split Computing","date":"2023-09-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/conditionally-adaptive-multi-task-learning","title":"Conditionally Adaptive Multi-Task Learning: Improving Transfer Learning in NLP Using Fewer Parameters & Less Data","date":"2020-09-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/smart-robust-and-efficient-fine-tuning-for","title":"SMART: Robust and Efficient Fine-Tuning for Pre-trained Natural Language Models through Principled Regularized Optimization","date":"2019-11-08","rows_on_this_dataset":5,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":6,"samples_unverified":2,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/simple-and-effective-text-matching-with-1","title":"Simple and Effective Text Matching with Richer Alignment Features","date":"2019-08-01","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multi-task-deep-neural-networks-for-natural","title":"Multi-Task Deep Neural Networks for Natural Language Understanding","date":"2019-01-31","rows_on_this_dataset":1,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":5,"samples_unverified":8,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/natural-language-inference-with-hierarchical","title":"Sentence Embeddings in NLI with Iterative Refinement Encoders","date":"2018-08-27","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/improving-language-understanding-by","title":"Improving Language Understanding by Generative Pre-Training","date":"2018-06-11","rows_on_this_dataset":1,"code_links":13,"syntology":null},{"paper":"/paper/compare-compress-and-propagate-enhancing","title":"Compare, Compress and Propagate: Enhancing Neural Architectures with Alignment Factorization for Natural Language Inference","date":"2017-12-30","rows_on_this_dataset":1,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":3,"samples_harvested":22,"samples_ran":12,"samples_unverified":10,"pointer_only_for_licence":3,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}