{"url":"/task/word-similarity","name":"Word Similarity","slug":"word-similarity","description_markdown":"Calculate a numerical score for the semantic similarity between two words.","categories":[{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":378,"papers_with_code":117,"benchmarks":1,"benchmark_tables_in_archive":1,"benchmark_tables_shown":1,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":3,"subtasks":0,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/word-similarity-on-ws353","slug":"word-similarity-on-ws353","dataset":"WS353","dataset_url":"/dataset/ws353","rows_in_archive":3,"metrics":["Spearman's Rho"],"first_row_in_archive_order":{"model":"Context-to-Vector","paper_title":"Using Context-to-Vector with Graph Retrofitting to Improve Word Embeddings","paper_url":"/paper/using-context-to-vector-with-graph-1","paper_date":"2022-10-30","arxiv_id":"2210.16848","code_links":[{"title":"binbinjiang/context2vector","url":"https://github.com/binbinjiang/context2vector"}],"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":2}}}],"datasets":[{"url":"/dataset/anlamver","name":"AnlamVer","full_name":"","num_papers_in_archive":4},{"url":"/dataset/ws353","name":"WS353","full_name":"WordSim-353","num_papers_in_archive":3},{"url":"/dataset/bangla-word-analogy-1","name":"Bangla Word Analogy","full_name":"Bangla Word Analogy","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":117,"tagged_in_all":378,"items":[{"url":"/paper/efficient-estimation-of-word-representations","title":"Efficient Estimation of Word Representations in Vector Space","date":"2013-01-16","arxiv_id":"1301.3781","repositories_listed":84,"syntology":{"n":62,"n_ran":19,"n_unverified":43,"n_pointer_only":18}},{"url":"/paper/enriching-word-vectors-with-subword","title":"Enriching Word Vectors with Subword Information","date":"2016-07-15","arxiv_id":"1607.04606","repositories_listed":54,"syntology":{"n":31,"n_ran":4,"n_unverified":27,"n_pointer_only":4}},{"url":"/paper/calculating-the-similarity-between-words-and","title":"Calculating the similarity between words and sentences using a lexical database and corpus statistics","date":"2018-02-15","arxiv_id":"1802.05667","repositories_listed":4,"syntology":null},{"url":"/paper/how-to-evaluate-word-embeddings-on-importance","title":"How to evaluate word embeddings? On importance of data efficiency and simple supervised tasks","date":"2017-02-07","arxiv_id":"1702.02170","repositories_listed":4,"syntology":null},{"url":"/paper/all-but-the-top-simple-and-effective","title":"All-but-the-Top: Simple and Effective Postprocessing for Word Representations","date":"2017-02-05","arxiv_id":"1702.01417","repositories_listed":4,"syntology":{"n":26,"n_ran":2,"n_unverified":24,"n_pointer_only":0}},{"url":"/paper/semglove-semantic-co-occurrences-for-glove","title":"SemGloVe: Semantic Co-occurrences for GloVe from BERT","date":"2020-12-30","arxiv_id":"2012.15197","repositories_listed":3,"syntology":null},{"url":"/paper/unsupervised-multilingual-word-embeddings","title":"Unsupervised Multilingual Word Embeddings","date":"2018-08-27","arxiv_id":"1808.08933","repositories_listed":3,"syntology":null},{"url":"/paper/speech2vec-a-sequence-to-sequence-framework","title":"Speech2Vec: A Sequence-to-Sequence Framework for Learning Word Embeddings from Speech","date":"2018-03-23","arxiv_id":"1803.08976","repositories_listed":3,"syntology":null},{"url":"/paper/visual-grounding-helps-learn-word-meanings-in","title":"Visual Grounding Helps Learn Word Meanings in Low-Data Regimes","date":"2023-10-20","arxiv_id":"2310.13257","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/can-a-fruit-fly-learn-word-embeddings-1","title":"Can a Fruit Fly Learn Word Embeddings?","date":"2021-01-18","arxiv_id":"2101.06887","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/a-common-semantic-space-for-monolingual-and","title":"A Common Semantic Space for Monolingual and Cross-Lingual Meta-Embeddings","date":"2020-01-17","arxiv_id":"2001.06381","repositories_listed":2,"syntology":null},{"url":"/paper/bcws-bilingual-contextual-word-similarity","title":"BCWS: Bilingual Contextual Word Similarity","date":"2018-10-21","arxiv_id":"1810.08951","repositories_listed":2,"syntology":null},{"url":"/paper/frage-frequency-agnostic-word-representation","title":"FRAGE: Frequency-Agnostic Word Representation","date":"2018-09-18","arxiv_id":"1809.06858","repositories_listed":2,"syntology":null},{"url":"/paper/skip-gram-word-embeddings-in-hyperbolic-space","title":"Skip-gram word embeddings in hyperbolic space","date":"2018-08-30","arxiv_id":"1809.01498","repositories_listed":2,"syntology":null},{"url":"/paper/learning-multilingual-word-embeddings-in","title":"Learning Multilingual Word Embeddings in Latent Metric Space: A Geometric Approach","date":"2018-08-27","arxiv_id":"1808.08773","repositories_listed":2,"syntology":null},{"url":"/paper/imparting-interpretability-to-word-embeddings","title":"Imparting Interpretability to Word Embeddings while Preserving Semantic Structure","date":"2018-07-19","arxiv_id":"1807.07279","repositories_listed":2,"syntology":null},{"url":"/paper/extrofitting-enriching-word-representation","title":"Extrofitting: Enriching Word Representation and its Vector Space with Semantic Lexicons","date":"2018-04-21","arxiv_id":"1804.07946","repositories_listed":2,"syntology":null},{"url":"/paper/a-simple-approach-to-learn-polysemous-word","title":"A Simple Approach to Learn Polysemous Word Embeddings","date":"2017-07-06","arxiv_id":"1707.01793","repositories_listed":2,"syntology":null},{"url":"/paper/multimodal-word-distributions","title":"Multimodal Word Distributions","date":"2017-04-27","arxiv_id":"1704.08424","repositories_listed":2,"syntology":null},{"url":"/paper/conceptnet-at-semeval-2017-task-2-extending","title":"ConceptNet at SemEval-2017 Task 2: Extending Word Embeddings with Multilingual Relational Knowledge","date":"2017-04-11","arxiv_id":"1704.03560","repositories_listed":2,"syntology":null},{"url":"/paper/construction-of-a-japanese-word-similarity","title":"Construction of a Japanese Word Similarity Dataset","date":"2017-03-17","arxiv_id":"1703.05916","repositories_listed":2,"syntology":null},{"url":"/paper/definition-modeling-learning-to-define-word","title":"Definition Modeling: Learning to define word embeddings in natural language","date":"2016-12-01","arxiv_id":"1612.00394","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/wordrank-learning-word-embeddings-via-robust","title":"WordRank: Learning Word Embeddings via Robust Ranking","date":"2015-06-09","arxiv_id":"1506.02761","repositories_listed":2,"syntology":null},{"url":"/paper/analyzing-continuous-semantic-shifts-with","title":"Analyzing Continuous Semantic Shifts with Diachronic Word Similarity Matrices","date":"2025-01-16","arxiv_id":"2501.09538","repositories_listed":1,"syntology":null},{"url":"/paper/lgde-local-graph-based-dictionary-expansion","title":"LGDE: Local Graph-based Dictionary Expansion","date":"2024-05-13","arxiv_id":"2405.07764","repositories_listed":1,"syntology":null},{"url":"/paper/a-likelihood-ratio-test-of-genetic","title":"A Likelihood Ratio Test of Genetic Relationship among Languages","date":"2024-03-30","arxiv_id":"2404.00284","repositories_listed":1,"syntology":null},{"url":"/paper/are-electra-s-sentence-embeddings-beyond","title":"Are ELECTRA's Sentence Embeddings Beyond Repair? The Case of Semantic Textual Similarity","date":"2024-02-20","arxiv_id":"2402.13130","repositories_listed":1,"syntology":null},{"url":"/paper/improving-heterogeneous-graph-learning-with","title":"Improving Heterogeneous Graph Learning with Weighted Mixed-Curvature Product Manifold","date":"2023-07-10","arxiv_id":"2307.04514","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":4}},{"url":"/paper/solving-cosine-similarity-underestimation","title":"Solving Cosine Similarity Underestimation between High Frequency Words by L2 Norm Discounting","date":"2023-05-17","arxiv_id":"2305.10610","repositories_listed":1,"syntology":null},{"url":"/paper/sanskritshala-a-neural-sanskrit-nlp-toolkit","title":"SanskritShala: A Neural Sanskrit NLP Toolkit with Web-Based Interface for Pedagogical and Annotation Purposes","date":"2023-02-19","arxiv_id":"2302.09527","repositories_listed":1,"syntology":null}],"syntology_records":7,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}