{"url":"/dataset/sts-benchmark","name":"STS Benchmark","full_name":"Semantic Textual Similarity","description_markdown":"STS Benchmark comprises a selection of the English datasets used in the STS tasks organized in the context of SemEval between 2012 and 2017. The selection of datasets include text from image captions, news headlines and user forums.\r\n\r\nSource: [STS Benchmark](http://ixa2.si.ehu.es/stswiki/index.php/STSbenchmark)","description_withheld":null,"homepage":"http://ixa2.si.ehu.es/stswiki/index.php/STSbenchmark","introduced_date":null,"introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[],"tasks":[{"name":"Semantic Textual Similarity","url":"/task/semantic-textual-similarity","datasets_with_task":"/datasets/task/semantic-textual-similarity"},{"name":"STS Benchmark","url":"/task/sts-benchmark","datasets_with_task":"/datasets/task/sts-benchmark"}],"languages":[],"variants":["STS Benchmark Dev","STS16","STS15","STS14","STS13","STS12","STS-B","STS Benchmark"],"data_loaders":[{"repo":"https://github.com/kamakshi-04/ML","url":"https://github.com/kamakshi-04/ML","frameworks":[]}],"num_papers_in_archive":45,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/semantic-textual-similarity-on-sts-benchmark","task":"Semantic Textual Similarity","dataset_variant":"STS Benchmark","rows":66,"metrics":["Pearson Correlation","Spearman Correlation","Accuracy","Dev Pearson Correlation","Dev Spearman Correlation"],"first_row_in_archive_order":{"model":"MT-DNN-SMART","paper":"/paper/smart-robust-and-efficient-fine-tuning-for","metrics":{"Pearson Correlation":"0.929","Spearman Correlation":"0.925"},"code_links":[{"title":"namisan/mt-dnn","url":"https://github.com/namisan/mt-dnn"},{"title":"microsoft/MT-DNN","url":"https://github.com/microsoft/MT-DNN"},{"title":"archinetai/smart-pytorch","url":"https://github.com/archinetai/smart-pytorch"},{"title":"archinetai/vat-pytorch","url":"https://github.com/archinetai/vat-pytorch"},{"title":"cliang1453/camero","url":"https://github.com/cliang1453/camero"},{"title":"chunhuililili/mt_dnn","url":"https://github.com/chunhuililili/mt_dnn"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/semantic-textual-similarity-on-sts13","task":"Semantic Textual Similarity","dataset_variant":"STS13","rows":22,"metrics":["Spearman Correlation"],"first_row_in_archive_order":{"model":"AnglE-LLaMA-7B","paper":"/paper/angle-optimized-text-embeddings","metrics":{"Spearman Correlation":"0.9058"},"code_links":[{"title":"SeanLee97/AnglE","url":"https://github.com/SeanLee97/AnglE"},{"title":"4ai/bellm","url":"https://github.com/4ai/bellm"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/semantic-textual-similarity-on-sts14","task":"Semantic Textual Similarity","dataset_variant":"STS14","rows":21,"metrics":["Spearman Correlation"],"first_row_in_archive_order":{"model":"AnglE-LLaMA-13B","paper":"/paper/angle-optimized-text-embeddings","metrics":{"Spearman Correlation":"0.8689"},"code_links":[{"title":"SeanLee97/AnglE","url":"https://github.com/SeanLee97/AnglE"},{"title":"4ai/bellm","url":"https://github.com/4ai/bellm"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/semantic-textual-similarity-on-sts12","task":"Semantic Textual Similarity","dataset_variant":"STS12","rows":20,"metrics":["Spearman Correlation"],"first_row_in_archive_order":{"model":"PromptEOL+CSE+OPT-13B","paper":"/paper/scaling-sentence-embeddings-with-large","metrics":{"Spearman Correlation":"0.8020"},"code_links":[{"title":"kongds/scaling_sentemb","url":"https://github.com/kongds/scaling_sentemb"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/semantic-textual-similarity-on-sts15","task":"Semantic Textual Similarity","dataset_variant":"STS15","rows":20,"metrics":["Spearman Correlation"],"first_row_in_archive_order":{"model":"PromptEOL+CSE+LLaMA-30B","paper":"/paper/scaling-sentence-embeddings-with-large","metrics":{"Spearman Correlation":"0.9004"},"code_links":[{"title":"kongds/scaling_sentemb","url":"https://github.com/kongds/scaling_sentemb"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/semantic-textual-similarity-on-sts16","task":"Semantic Textual Similarity","dataset_variant":"STS16","rows":20,"metrics":["Spearman Correlation"],"first_row_in_archive_order":{"model":"AnglE-LLaMA-7B-v2","paper":"/paper/angle-optimized-text-embeddings","metrics":{"Spearman Correlation":"0.8700"},"code_links":[{"title":"SeanLee97/AnglE","url":"https://github.com/SeanLee97/AnglE"},{"title":"4ai/bellm","url":"https://github.com/4ai/bellm"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/rematch-robust-and-efficient-matching-of","title":"Rematch: Robust and Efficient Matching of Local Knowledge Graphs to Improve Structural and Semantic Similarity","date":"2024-04-02","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/def2vec-extensible-word-embeddings-from","title":"Def2Vec: Extensible Word Embeddings from Dictionary Definitions","date":"2023-12-16","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/angle-optimized-text-embeddings","title":"AnglE-optimized Text Embeddings","date":"2023-09-22","rows_on_this_dataset":15,"code_links":2,"syntology":null},{"paper":"/paper/scaling-sentence-embeddings-with-large","title":"Scaling Sentence Embeddings with Large Language Models","date":"2023-07-31","rows_on_this_dataset":18,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":4,"samples_unverified":0,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/llm-int8-8-bit-matrix-multiplication-for","title":"LLM.int8(): 8-bit Matrix Multiplication for Transformers at Scale","date":"2022-08-15","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":2,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/adversarial-self-attention-for-language","title":"Adversarial Self-Attention for Language Understanding","date":"2022-06-25","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/diffcse-difference-based-contrastive-learning","title":"DiffCSE: Difference-based Contrastive Learning for Sentence Embeddings","date":"2022-04-21","rows_on_this_dataset":10,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":2,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/deep-continuous-prompt-for-contrastive-1","title":"Improved Universal Sentence Embeddings with Prompt-based Contrastive Learning and Energy-based Learning","date":"2022-03-14","rows_on_this_dataset":6,"code_links":1,"syntology":null},{"paper":"/paper/mnet-sim-a-multi-layered-semantic-similarity-1","title":"MNet-Sim: A Multi-layered Semantic Similarity Network to Evaluate Sentence Similarity","date":"2021-11-09","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/trans-encoder-unsupervised-sentence-pair","title":"Trans-Encoder: Unsupervised sentence-pair modelling through self- and mutual-distillations","date":"2021-09-27","rows_on_this_dataset":28,"code_links":1,"syntology":null},{"paper":"/paper/charformer-fast-character-transformers-via","title":"Charformer: Fast Character Transformers via Gradient-based Subword Tokenization","date":"2021-06-23","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":7,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/fnet-mixing-tokens-with-fourier-transforms","title":"FNet: Mixing Tokens with Fourier Transforms","date":"2021-05-09","rows_on_this_dataset":1,"code_links":12,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/entailment-as-few-shot-learner","title":"Entailment as Few-Shot Learner","date":"2021-04-29","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/simcse-simple-contrastive-learning-of","title":"SimCSE: Simple Contrastive Learning of Sentence Embeddings","date":"2021-04-18","rows_on_this_dataset":9,"code_links":23,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":30,"samples_ran":17,"samples_unverified":13,"pointer_only_for_licence":19,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/fast-effective-and-self-supervised","title":"Fast, Effective, and Self-Supervised: Transforming Masked Language Models into Universal Lexical and Sentence Encoders","date":"2021-04-16","rows_on_this_dataset":12,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/how-to-train-bert-with-an-academic-budget","title":"How to Train BERT with an Academic Budget","date":"2021-04-15","rows_on_this_dataset":1,"code_links":4,"syntology":null},{"paper":"/paper/generating-datasets-with-pretrained-language","title":"Generating Datasets with Pretrained Language Models","date":"2021-04-15","rows_on_this_dataset":7,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":0,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/clear-contrastive-learning-for-sentence","title":"CLEAR: Contrastive Learning for Sentence Representation","date":"2020-12-31","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/informer-transformer-likes-informed-attention","title":"RealFormer: Transformer Likes Residual Attention","date":"2020-12-21","rows_on_this_dataset":1,"code_links":5,"syntology":null},{"paper":"/paper/on-the-sentence-embeddings-from-pre-trained","title":"On the Sentence Embeddings from Pre-trained Language Models","date":"2020-11-02","rows_on_this_dataset":6,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/a-statistical-framework-for-low-bitwidth","title":"A Statistical Framework for Low-bitwidth Training of Deep Neural Networks","date":"2020-10-27","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":1,"samples_unverified":3,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/an-unsupervised-sentence-embedding-method","title":"An Unsupervised Sentence Embedding Method by Mutual Information Maximization","date":"2020-09-25","rows_on_this_dataset":6,"code_links":1,"syntology":null},{"paper":"/paper/big-bird-transformers-for-longer-sequences","title":"Big Bird: Transformers for Longer Sequences","date":"2020-07-28","rows_on_this_dataset":1,"code_links":14,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":10,"samples_unverified":5,"pointer_only_for_licence":11,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/deberta-decoding-enhanced-bert-with","title":"DeBERTa: Decoding-enhanced BERT with Disentangled Attention","date":"2020-06-05","rows_on_this_dataset":1,"code_links":14,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":4,"samples_unverified":9,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/smart-robust-and-efficient-fine-tuning-for","title":"SMART: Robust and Efficient Fine-Tuning for Pre-trained Natural Language Models through Principled Regularized Optimization","date":"2019-11-08","rows_on_this_dataset":3,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":6,"samples_unverified":2,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/exploring-the-limits-of-transfer-learning","title":"Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer","date":"2019-10-23","rows_on_this_dataset":6,"code_links":57,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":31,"samples_ran":2,"samples_unverified":29,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/q8bert-quantized-8bit-bert","title":"Q8BERT: Quantized 8Bit BERT","date":"2019-10-14","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":3,"samples_unverified":8,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/distilbert-a-distilled-version-of-bert","title":"DistilBERT, a distilled version of BERT: smaller, faster, cheaper and lighter","date":"2019-10-02","rows_on_this_dataset":1,"code_links":37,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":27,"samples_ran":19,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/albert-a-lite-bert-for-self-supervised","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","date":"2019-09-26","rows_on_this_dataset":1,"code_links":48,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":126,"samples_ran":46,"samples_unverified":80,"pointer_only_for_licence":22,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/190910351","title":"TinyBERT: Distilling BERT for Natural Language Understanding","date":"2019-09-23","rows_on_this_dataset":1,"code_links":10,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/q-bert-hessian-based-ultra-low-precision","title":"Q-BERT: Hessian Based Ultra Low Precision Quantization of BERT","date":"2019-09-12","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/sentence-bert-sentence-embeddings-using","title":"Sentence-BERT: Sentence Embeddings using Siamese BERT-Networks","date":"2019-08-27","rows_on_this_dataset":11,"code_links":64,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":58,"samples_ran":20,"samples_unverified":38,"pointer_only_for_licence":11,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/structbert-incorporating-language-structures","title":"StructBERT: Incorporating Language Structures into Pre-training for Deep Language Understanding","date":"2019-08-13","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/ernie-20-a-continual-pre-training-framework","title":"ERNIE 2.0: A Continual Pre-training Framework for Language Understanding","date":"2019-07-29","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/roberta-a-robustly-optimized-bert-pretraining","title":"RoBERTa: A Robustly Optimized BERT Pretraining Approach","date":"2019-07-26","rows_on_this_dataset":1,"code_links":67,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":48,"samples_ran":22,"samples_unverified":26,"pointer_only_for_licence":23,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/spanbert-improving-pre-training-by","title":"SpanBERT: Improving Pre-training by Representing and Predicting Spans","date":"2019-07-24","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":3,"samples_unverified":12,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/xlnet-generalized-autoregressive-pretraining","title":"XLNet: Generalized Autoregressive Pretraining for Language Understanding","date":"2019-06-19","rows_on_this_dataset":1,"code_links":27,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":24,"samples_ran":10,"samples_unverified":14,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ernie-enhanced-language-representation-with","title":"ERNIE: Enhanced Language Representation with Informative Entities","date":"2019-05-17","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/bert-pre-training-of-deep-bidirectional","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","date":"2018-10-11","rows_on_this_dataset":1,"code_links":534,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":659,"samples_ran":204,"samples_unverified":455,"pointer_only_for_licence":149,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/universal-sentence-encoder","title":"Universal Sentence Encoder","date":"2018-03-29","rows_on_this_dataset":1,"code_links":24,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":22,"samples_ran":1,"samples_unverified":21,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":27,"samples_harvested":1134,"samples_ran":392,"samples_unverified":742,"pointer_only_for_licence":262,"papers_with_no_sample_that_ran":4,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}