{"url":"/method/augmented-sbert","slug":"augmented-sbert","name":"Augmented SBERT","full_name":"Augmented SBERT","full_name_withheld":false,"description_markdown":"**Augmented SBERT** is a data augmentation strategy for pairwise sentence scoring that uses a [BERT](https://paperswithcode.com/method/bert) cross-encoder to improve the performance for the [SBERT](https://paperswithcode.com/method/sbert) bi-encoders. Given a pre-trained, well-performing crossencoder, we sample sentence pairs according to a certain sampling strategy and label these using the cross-encoder. We call these weakly labeled examples the silver dataset and they will be merged with the gold training dataset. We then train the bi-encoder on this extended training dataset.","description_state":"present","introduced_year":null,"introduced_by":{"title":"Augmented SBERT: Data Augmentation Method for Improving Bi-Encoders for Pairwise Sentence Scoring Tasks","paper":"/paper/augmented-sbert-data-augmentation-method-for","first_author":"Nandan Thakur","n_authors":4,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/augmented-sbert-data-augmentation-method-for"},"source":{"url":"https://arxiv.org/abs/2010.08240v2","title":"Augmented SBERT: Data Augmentation Method for Improving Bi-Encoders for Pairwise Sentence Scoring Tasks","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Text Augmentation","url":"/methods/category/text-augmentation","pwc_aliases":[]}],"n_papers_tagged":1,"archive_num_papers":1,"papers_newest_first":[{"paper":"/paper/augmented-sbert-data-augmentation-method-for","title":"Augmented SBERT: Data Augmentation Method for Improving Bi-Encoders for Pairwise Sentence Scoring Tasks","date":"2020-10-16","arxiv_id":"2010.08240","n_code_links":1,"syntology":null}],"papers_shown":1,"tasks":[{"task":"/task/data-augmentation","name":"Data Augmentation","papers":1},{"task":"/task/domain-adaptation","name":"Domain Adaptation","papers":1},{"task":"/task/paraphrase-identification-within-bi-encoder","name":"Paraphrase Identification within Bi-Encoder","papers":1},{"task":"/task/semantic-textual-similarity","name":"Semantic Textual Similarity","papers":1},{"task":"/task/semantic-textual-similarity-within-bi-encoder","name":"Semantic Textual Similarity within Bi-Encoder","papers":1},{"task":"/task/sentence","name":"Sentence","papers":1},{"task":"/task/sentence-pair-modeling","name":"Sentence Pair Modeling","papers":1}],"tasks_shown":7,"n_tasks":7,"usage_by_year":[{"year":"2020","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/augmented-sbert"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}