{"url":"/dataset/glue","name":"GLUE","full_name":"General Language Understanding Evaluation benchmark","description_markdown":"General Language Understanding Evaluation (**GLUE**) benchmark is a collection of nine natural language understanding tasks, including single-sentence tasks CoLA and SST-2, similarity and paraphrasing tasks MRPC, STS-B and QQP, and natural language inference tasks MNLI, QNLI, RTE and WNLI.\r\n\r\nSource: [Align, Mask and Select: A Simple Method for Incorporating Commonsense Knowledge into Language Representation Models](https://arxiv.org/abs/1908.06725)\r\nImage Source: [https://gluebenchmark.com/](https://gluebenchmark.com/)","description_withheld":null,"homepage":"https://gluebenchmark.com/","introduced_date":"2019-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/glue-a-multi-task-benchmark-and-analysis","title":"GLUE: A Multi-Task Benchmark and Analysis Platform for Natural Language Understanding","first_author":"Alex Wang","url":null},"license":{"name":"Custom (various)","url":"https://gluebenchmark.com/faq"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Classification","url":"/task/classification-1","datasets_with_task":"/datasets/task/classification-1"},{"name":"Text Classification","url":"/task/text-classification","datasets_with_task":"/datasets/task/text-classification"},{"name":"Text Generation","url":"/task/text-generation","datasets_with_task":"/datasets/task/text-generation"},{"name":"Few-Shot Learning","url":"/task/few-shot-learning","datasets_with_task":"/datasets/task/few-shot-learning"},{"name":"Natural Language Inference","url":"/task/natural-language-inference","datasets_with_task":"/datasets/task/natural-language-inference"},{"name":"Semantic Textual Similarity","url":"/task/semantic-textual-similarity","datasets_with_task":"/datasets/task/semantic-textual-similarity"},{"name":"Stochastic Optimization","url":"/task/stochastic-optimization","datasets_with_task":"/datasets/task/stochastic-optimization"},{"name":"Natural Language Understanding","url":"/task/natural-language-understanding","datasets_with_task":"/datasets/task/natural-language-understanding"},{"name":"Linguistic Acceptability","url":"/task/linguistic-acceptability","datasets_with_task":"/datasets/task/linguistic-acceptability"},{"name":"Data-free Knowledge Distillation","url":"/task/data-free-knowledge-distillation","datasets_with_task":"/datasets/task/data-free-knowledge-distillation"},{"name":"Model Compression","url":"/task/model-compression","datasets_with_task":"/datasets/task/model-compression"},{"name":"Semantic Textual Similarity within Bi-Encoder","url":"/task/semantic-textual-similarity-within-bi-encoder","datasets_with_task":"/datasets/task/semantic-textual-similarity-within-bi-encoder"},{"name":"Sentence-Embedding","url":"/task/sentence-embedding-1","datasets_with_task":"/datasets/task/sentence-embedding-1"},{"name":"QQP","url":"/task/qqp","datasets_with_task":"/datasets/task/qqp"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["WNLI","RTE","QNLI","MNLI-mm","MNLI-m","qqp","STS-B","MRPC","SST-2","CoLA","FinanceInc/auditor_sentiment","CHANGE-IT","datasetX","GLUE QNLI Dev","GLUE SST2 Dev","GLUE STSB","GLUE SST2","GLUE RTE","GLUE QQP","GLUE QNLI","GLUE MNLI","GLUE COLA","GLUE WNLI","GLUE MRPC","GLUE"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/nyu-mll/glue","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/glue","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/asoria/glue-test","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/polinaeterna/glue","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/gsarti/change_it","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/asoria/glue","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/ncats/EpiSet4BinaryClassification","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/vector/structuretest","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/evaluate/glue-ci","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/quincyqiang/test","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/severo/glue","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/truezhichu/ttt","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/mariosasko/glue","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/gorinars/test","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/facebookresearch/ParlAI","url":"https://parl.ai/docs/tasks.html#glue","frameworks":["pytorch"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/glue","frameworks":["tf","jax"]}],"num_papers_in_archive":3197,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/natural-language-understanding-on-glue","task":"Natural Language Understanding","dataset_variant":"GLUE","rows":2,"metrics":["Average"],"first_row_in_archive_order":{"model":"MT-DNN-SMART","paper":"/paper/smart-robust-and-efficient-fine-tuning-for","metrics":{"Average":"89.9"},"code_links":[{"title":"namisan/mt-dnn","url":"https://github.com/namisan/mt-dnn"},{"title":"microsoft/MT-DNN","url":"https://github.com/microsoft/MT-DNN"},{"title":"archinetai/smart-pytorch","url":"https://github.com/archinetai/smart-pytorch"},{"title":"archinetai/vat-pytorch","url":"https://github.com/archinetai/vat-pytorch"},{"title":"cliang1453/camero","url":"https://github.com/cliang1453/camero"},{"title":"chunhuililili/mt_dnn","url":"https://github.com/chunhuililili/mt_dnn"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-classification-on-glue-sst2","task":"Text Classification","dataset_variant":"GLUE SST2","rows":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"TRANS-BLSTM","paper":"/paper/trans-blstm-transformer-with-bidirectional","metrics":{"Accuracy":"94.38"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/few-shot-learning-on-glue-qqp","task":"Few-Shot Learning","dataset_variant":"GLUE QQP","rows":1,"metrics":["F1-score"],"first_row_in_archive_order":{"model":"DART","paper":"/paper/differentiable-prompt-makes-pre-trained","metrics":{"F1-score":"67.8(3.2)"},"code_links":[{"title":"zjunlp/DART","url":"https://github.com/zjunlp/DART"},{"title":"zhengxiangshi/powerfulpromptft","url":"https://github.com/zhengxiangshi/powerfulpromptft"},{"title":"paperspapers/badprompt","url":"https://github.com/paperspapers/badprompt"},{"title":"zhaohan-xi/plm-prompt-defense","url":"https://github.com/zhaohan-xi/plm-prompt-defense"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-classification-on-glue-cola","task":"Text Classification","dataset_variant":"GLUE COLA","rows":1,"metrics":["Matthews Correlation"],"first_row_in_archive_order":{"model":"TRANS-BLSTM","paper":"/paper/trans-blstm-transformer-with-bidirectional","metrics":{},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-classification-on-glue-mrpc","task":"Text Classification","dataset_variant":"GLUE MRPC","rows":1,"metrics":["Accuracy","F1"],"first_row_in_archive_order":{"model":"TRANS-BLSTM","paper":"/paper/trans-blstm-transformer-with-bidirectional","metrics":{"Accuracy":"90.45"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-classification-on-glue-rte","task":"Text Classification","dataset_variant":"GLUE RTE","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"TRANS-BLSTM","paper":"/paper/trans-blstm-transformer-with-bidirectional","metrics":{"Accuracy":"79.78"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-classification-on-glue-stsb","task":"Text Classification","dataset_variant":"GLUE STSB","rows":1,"metrics":["Spearmanr"],"first_row_in_archive_order":{"model":"TRANS-BLSTM","paper":"/paper/trans-blstm-transformer-with-bidirectional","metrics":{},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/natural-language-inference-on-glue","task":"Natural Language Inference","dataset_variant":"GLUE","rows":0,"metrics":["Accuracy","F1 Macro","F1 Micro","F1 Weighted","Precision Macro","Precision Micro","Precision Weighted","Recall Macro","Recall Micro","Recall Weighted","loss"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-classification-on-glue","task":"Text Classification","dataset_variant":"GLUE","rows":0,"metrics":["AUC","Accuracy","F1","Matthews Correlation","Precision","Recall","loss"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-classification-on-glue-mnli","task":"Text Classification","dataset_variant":"GLUE MNLI","rows":0,"metrics":["Accuracy"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-classification-on-glue-qnli","task":"Text Classification","dataset_variant":"GLUE QNLI","rows":0,"metrics":["Accuracy"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-classification-on-glue-qqp","task":"Text Classification","dataset_variant":"GLUE QQP","rows":0,"metrics":["Accuracy","F1"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-classification-on-glue-wnli","task":"Text Classification","dataset_variant":"GLUE WNLI","rows":0,"metrics":["Accuracy"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/guidelines-for-the-regularization-of-gammas","title":"Guidelines for the Regularization of Gammas in Batch Normalization for Deep Residual Networks","date":"2022-05-15","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/differentiable-prompt-makes-pre-trained","title":"Differentiable Prompt Makes Pre-trained Language Models Better Few-shot Learners","date":"2021-08-30","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/trans-blstm-transformer-with-bidirectional","title":"TRANS-BLSTM: Transformer with Bidirectional LSTM for Language Understanding","date":"2020-03-16","rows_on_this_dataset":5,"code_links":0,"syntology":null},{"paper":"/paper/smart-robust-and-efficient-fine-tuning-for","title":"SMART: Robust and Efficient Fine-Tuning for Pre-trained Natural Language Models through Principled Regularized Optimization","date":"2019-11-08","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":8,"samples_ran":8,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/bert-pre-training-of-deep-bidirectional","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","date":"2018-10-11","rows_on_this_dataset":1,"code_links":534,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":659,"samples_ran":208,"samples_unverified":451,"pointer_only_for_licence":149,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":3,"samples_harvested":671,"samples_ran":219,"samples_unverified":452,"pointer_only_for_licence":150,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}