{"url":"/task/sentence-similarity","name":"Sentence Similarity","slug":"sentence-similarity","description_markdown":null,"categories":[],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":194,"papers_with_code":76,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":1,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[{"url":"/dataset/biosses","name":"BIOSSES","full_name":"Biomedical Semantic Similarity Estimation System","num_papers_in_archive":38}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":76,"tagged_in_all":194,"items":[{"url":"/paper/senteval-an-evaluation-toolkit-for-universal","title":"SentEval: An Evaluation Toolkit for Universal Sentence Representations","date":"2018-03-14","arxiv_id":"1803.05449","repositories_listed":11,"syntology":{"n":8,"n_ran":6,"n_unverified":2,"n_pointer_only":6}},{"url":"/paper/poor-man-s-bert-smaller-and-faster","title":"On the Effect of Dropping Layers of Pre-trained Transformer Models","date":"2020-04-08","arxiv_id":"2004.03844","repositories_listed":4,"syntology":{"n":8,"n_ran":4,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/calculating-the-similarity-between-words-and","title":"Calculating the similarity between words and sentences using a lexical database and corpus statistics","date":"2018-02-15","arxiv_id":"1802.05667","repositories_listed":4,"syntology":null},{"url":"/paper/generating-sentences-by-editing-prototypes","title":"Generating Sentences by Editing Prototypes","date":"2017-09-26","arxiv_id":"1709.08878","repositories_listed":3,"syntology":null},{"url":"/paper/contrastive-learning-of-sentence-embeddings","title":"Contrastive Learning of Sentence Embeddings from Scratch","date":"2023-05-24","arxiv_id":"2305.15077","repositories_listed":2,"syntology":{"n":6,"n_ran":2,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/domain-specific-language-model-pretraining","title":"Domain-Specific Language Model Pretraining for Biomedical Natural Language Processing","date":"2020-07-31","arxiv_id":"2007.15779","repositories_listed":2,"syntology":null},{"url":"/paper/neuralwarp-time-series-similarity-with","title":"NeuralWarp: Time-Series Similarity with Warping Networks","date":"2018-12-20","arxiv_id":"1812.08306","repositories_listed":2,"syntology":null},{"url":"/paper/context-movers-distance-barycenters-optimal","title":"Context Mover's Distance & Barycenters: Optimal Transport of Contexts for Building Representations","date":"2018-08-29","arxiv_id":"1808.09663","repositories_listed":2,"syntology":null},{"url":"/paper/macro-grammars-and-holistic-triggering-for","title":"Macro Grammars and Holistic Triggering for Efficient Semantic Parsing","date":"2017-07-25","arxiv_id":"1707.07806","repositories_listed":2,"syntology":null},{"url":"/paper/sentence-ordering-and-coherence-modeling","title":"Sentence Ordering and Coherence Modeling using Recurrent Neural Networks","date":"2016-11-08","arxiv_id":"1611.02654","repositories_listed":2,"syntology":null},{"url":"/paper/can-linguists-better-understand-dna","title":"Can linguists better understand DNA?","date":"2024-12-10","arxiv_id":"2412.07678","repositories_listed":1,"syntology":null},{"url":"/paper/a-novel-word-pair-based-gaussian-sentence","title":"A Novel Word Pair-based Gaussian Sentence Similarity Algorithm For Bengali Extractive Text Summarization","date":"2024-11-26","arxiv_id":"2411.17181","repositories_listed":1,"syntology":null},{"url":"/paper/toeing-the-party-line-election-manifestos-as","title":"Toeing the Party Line: Election Manifestos as a Key to Understand Political Discourse on Twitter","date":"2024-10-21","arxiv_id":"2410.15743","repositories_listed":1,"syntology":null},{"url":"/paper/word-embedding-dimension-reduction-via-weakly","title":"Word Embedding Dimension Reduction via Weakly-Supervised Feature Selection","date":"2024-07-17","arxiv_id":"2407.12342","repositories_listed":1,"syntology":null},{"url":"/paper/supergleber-german-language-understanding","title":"SuperGLEBer: German Language Understanding Evaluation Benchmark","date":"2024-06-20","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/ottawa-optimal-transport-adaptive-word","title":"OTTAWA: Optimal TransporT Adaptive Word Aligner for Hallucination and Omission Translation Errors Detection","date":"2024-06-04","arxiv_id":"2406.01919","repositories_listed":1,"syntology":null},{"url":"/paper/from-news-to-summaries-building-a-hungarian","title":"From News to Summaries: Building a Hungarian Corpus for Extractive and Abstractive Summarization","date":"2024-04-04","arxiv_id":"2404.03555","repositories_listed":1,"syntology":null},{"url":"/paper/contrastive-learning-and-mixture-of-experts","title":"Contrastive Learning and Mixture of Experts Enables Precise Vector Embeddings","date":"2024-01-28","arxiv_id":"2401.15713","repositories_listed":1,"syntology":null},{"url":"/paper/gloss-attention-for-gloss-free-sign-language-1","title":"Gloss Attention for Gloss-free Sign Language Translation","date":"2023-07-14","arxiv_id":"2307.07361","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/lea-improving-sentence-similarity-robustness","title":"LEA: Improving Sentence Similarity Robustness to Typos Using Lexical Attention Bias","date":"2023-07-06","arxiv_id":"2307.02912","repositories_listed":1,"syntology":null},{"url":"/paper/what-do-self-supervised-speech-models-know","title":"What Do Self-Supervised Speech Models Know About Words?","date":"2023-06-30","arxiv_id":"2307.00162","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-anisotropy-and-outliers-in","title":"Exploring Anisotropy and Outliers in Multilingual Language Models for Cross-Lingual Semantic Sentence Similarity","date":"2023-06-01","arxiv_id":"2306.00458","repositories_listed":1,"syntology":null},{"url":"/paper/csts-conditional-semantic-textual-similarity","title":"C-STS: Conditional Semantic Textual Similarity","date":"2023-05-24","arxiv_id":"2305.15093","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_unverified":2,"n_pointer_only":4}},{"url":"/paper/imad-image-augmented-multi-modal-dialogue","title":"IMAD: IMage-Augmented multi-modal Dialogue","date":"2023-05-17","arxiv_id":"2305.10512","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/improving-sentence-similarity-estimation-for","title":"Improving Sentence Similarity Estimation for Unsupervised Extractive Summarization","date":"2023-02-24","arxiv_id":"2302.12490","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-exploit-temporal-structure-for","title":"Learning to Exploit Temporal Structure for Biomedical Vision-Language Processing","date":"2023-01-11","arxiv_id":"2301.04558","repositories_listed":1,"syntology":null},{"url":"/paper/on-isotropy-and-learning-dynamics-of","title":"On Isotropy, Contextualization and Learning Dynamics of Contrastive-based Sentence Representation Learning","date":"2022-12-18","arxiv_id":"2212.09170","repositories_listed":1,"syntology":null},{"url":"/paper/l3cube-mahasbert-and-hindsbert-sentence-bert","title":"L3Cube-MahaSBERT and HindSBERT: Sentence BERT Models and Benchmarking BERT Sentence Representations for Hindi and Marathi","date":"2022-11-21","arxiv_id":"2211.11187","repositories_listed":1,"syntology":null},{"url":"/paper/subspace-based-set-operations-on-a-pre","title":"Subspace Representations for Soft Set Operations and Sentence Similarities","date":"2022-10-24","arxiv_id":"2210.13034","repositories_listed":1,"syntology":null},{"url":"/paper/reweighting-strategy-based-on-synthetic-data","title":"Reweighting Strategy based on Synthetic Data Identification for Sentence Similarity","date":"2022-08-29","arxiv_id":"2208.13376","repositories_listed":1,"syntology":null}],"syntology_records":6,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}