{"url":"/dataset/senteval","name":"SentEval","full_name":null,"description_markdown":"SentEval is a toolkit for evaluating the quality of universal sentence representations. SentEval encompasses a variety of tasks, including binary and multi-class classification, natural language inference and sentence similarity. The set of tasks was selected based on what appears to be the community consensus regarding the appropriate evaluations for universal sentence representations. The toolkit comes with scripts to download and preprocess datasets, and an easy interface to evaluate sentence encoders.\r\n\r\nSource: [SentEval: An Evaluation Toolkit for Universal Sentence Representations](/paper/senteval-an-evaluation-toolkit-for-universal)","description_withheld":null,"homepage":"https://arxiv.org/abs/1803.05449","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/senteval-an-evaluation-toolkit-for-universal","title":"SentEval: An Evaluation Toolkit for Universal Sentence Representations","first_author":"Alexis Conneau","url":null},"license":{"name":"Various","url":"https://github.com/facebookresearch/SentEval"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Semantic Textual Similarity","url":"/task/semantic-textual-similarity","datasets_with_task":"/datasets/task/semantic-textual-similarity"},{"name":"Linear-Probe Classification","url":"/task/linear-probe-classification","datasets_with_task":"/datasets/task/linear-probe-classification"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["SentEval"],"data_loaders":[],"num_papers_in_archive":168,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/semantic-textual-similarity-on-senteval","task":"Semantic Textual Similarity","dataset_variant":"SentEval","rows":6,"metrics":["MRPC","SICK-R","SICK-E","STS"],"first_row_in_archive_order":{"model":"GenSen","paper":"/paper/learning-general-purpose-distributed-sentence","metrics":{"MRPC":"78.6/84.4","SICK-E":"87.8","SICK-R":"0.888","STS":"78.9/78.6"},"code_links":[{"title":"facebookresearch/InferSent","url":"https://github.com/facebookresearch/InferSent"},{"title":"facebookresearch/SentEval","url":"https://github.com/facebookresearch/SentEval"},{"title":"Maluuba/gensen","url":"https://github.com/Maluuba/gensen"},{"title":"najafmurtaza/Developing-Machine-Learning-Models-in-Flask","url":"https://github.com/najafmurtaza/Developing-Machine-Learning-Models-in-Flask"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/xlnet-generalized-autoregressive-pretraining","title":"XLNet: Generalized Autoregressive Pretraining for Language Understanding","date":"2019-06-19","rows_on_this_dataset":1,"code_links":27,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":24,"samples_ran":10,"samples_unverified":14,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/improving-multi-task-deep-neural-networks-via","title":"Improving Multi-Task Deep Neural Networks via Knowledge Distillation for Natural Language Understanding","date":"2019-04-20","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/training-complex-models-with-multi-task-weak","title":"Training Complex Models with Multi-Task Weak Supervision","date":"2018-10-05","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/learning-general-purpose-distributed-sentence","title":"Learning General Purpose Distributed Sentence Representations via Large Scale Multi-task Learning","date":"2018-03-30","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/supervised-learning-of-universal-sentence","title":"Supervised Learning of Universal Sentence Representations from Natural Language Inference Data","date":"2017-05-05","rows_on_this_dataset":1,"code_links":23,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":6,"samples_unverified":1,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/discriminative-improvements-to-distributional","title":"Discriminative Improvements to Distributional Sentence Similarity","date":"2013-10-01","rows_on_this_dataset":1,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":3,"samples_harvested":34,"samples_ran":19,"samples_unverified":15,"pointer_only_for_licence":13,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}