{"url":"/method/skip-gram-word2vec","slug":"skip-gram-word2vec","name":"Skip-gram Word2Vec","full_name":"Skip-gram Word2Vec","full_name_withheld":false,"description_markdown":"**Skip-gram Word2Vec** is an architecture for computing word embeddings. Instead of using surrounding words to predict the center word, as with CBow Word2Vec, Skip-gram Word2Vec uses the central word to predict the surrounding words.\r\n\r\nThe skip-gram objective function sums the log probabilities of the surrounding $n$ words to the left and right of the target word $w\\_{t}$ to produce the following objective:\r\n\r\n$$J\\_\\theta = \\frac{1}{T}\\sum^{T}\\_{t=1}\\sum\\_{-n\\leq{j}\\leq{n}, \\neq{0}}\\log{p}\\left(w\\_{j+1}\\mid{w\\_{t}}\\right)$$","description_state":"present","introduced_year":null,"introduced_by":{"title":"Efficient Estimation of Word Representations in Vector Space","paper":"/paper/efficient-estimation-of-word-representations","first_author":"Tomas Mikolov","n_authors":4,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/efficient-estimation-of-word-representations"},"source":{"url":"http://arxiv.org/abs/1301.3781v3","title":"Efficient Estimation of Word Representations in Vector Space","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Static Word Embeddings","url":"/methods/category/static-word-embeddings","pwc_aliases":[]},{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Word Embeddings","url":"/methods/category/word-embeddings","pwc_aliases":[]}],"n_papers_tagged":14,"archive_num_papers":14,"papers_newest_first":[{"paper":"/paper/rdf-star2vec-rdf-star-graph-embeddings-for","title":"RDF-star2Vec: RDF-star Graph Embeddings for Data Mining","date":"2023-12-25","arxiv_id":"2312.15626","n_code_links":2,"syntology":null},{"paper":"/paper/evaluating-unsupervised-text-classification","title":"Evaluating Unsupervised Text Classification: Zero-shot and Similarity-based Approaches","date":"2022-11-29","arxiv_id":"2211.16285","n_code_links":2,"syntology":null},{"paper":null,"title":"Walk this Way! Entity Walks and Property Walks for RDF2vec","date":"2022-04-05","arxiv_id":"2204.02777","n_code_links":0,"syntology":null},{"paper":null,"title":"Enhancing Sindhi Word Segmentation using Subword Representation Learning and Position-aware Self-attention","date":"2020-12-30","arxiv_id":"2012.15079","n_code_links":0,"syntology":null},{"paper":"/paper/improving-clinical-document-understanding-on","title":"Improving Clinical Document Understanding on COVID-19 Research with Spark NLP","date":"2020-12-07","arxiv_id":"2012.04005","n_code_links":1,"syntology":null},{"paper":null,"title":"Transformer Query-Target Knowledge Discovery (TEND): Drug Discovery from CORD-19","date":"2020-11-28","arxiv_id":"2012.04682","n_code_links":0,"syntology":null},{"paper":"/paper/farstail-a-persian-natural-language-inference","title":"FarsTail: A Persian Natural Language Inference Dataset","date":"2020-09-18","arxiv_id":"2009.08820","n_code_links":1,"syntology":null},{"paper":null,"title":"Interpreting Pretrained Contextualized Representations via Reductions to Static Embeddings","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"title":"Why is penguin more similar to polar bear than to sea gull? Analyzing conceptual knowledge in distributional models","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/topic-modeling-in-embedding-spaces","title":"Topic Modeling in Embedding Spaces","date":"2019-07-08","arxiv_id":"1907.04907","n_code_links":12,"syntology":{"ran":3,"of":13,"unverified":10,"pointer_only":0}},{"paper":null,"title":"An Empirical Evaluation of Text Representation Schemes on Multilingual Social Web to Filter the Textual Aggression","date":"2019-04-16","arxiv_id":"1904.08770","n_code_links":0,"syntology":null},{"paper":null,"title":"Predicting Role Relevance with Minimal Domain Expertise in a Financial Domain","date":"2017-04-19","arxiv_id":"1704.05571","n_code_links":0,"syntology":null},{"paper":null,"title":"Effective search space reduction for spell correction using character neural embeddings","date":"2017-04-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/efficient-estimation-of-word-representations","title":"Efficient Estimation of Word Representations in Vector Space","date":"2013-01-16","arxiv_id":"1301.3781","n_code_links":84,"syntology":{"ran":19,"of":62,"unverified":43,"pointer_only":18}}],"papers_shown":14,"tasks":[{"task":"/task/word-embeddings","name":"Word Embeddings","papers":3},{"task":"/task/graph-embedding","name":"Graph Embedding","papers":2},{"task":"/task/knowledge-graph-embedding","name":"Knowledge Graph Embedding","papers":2},{"task":"/task/knowledge-graphs","name":"Knowledge Graphs","papers":2},{"task":"/task/anatomy","name":"Anatomy","papers":1},{"task":"/task/benchmarking","name":"Benchmarking","papers":1},{"task":"/task/classification-1","name":"Classification","papers":1},{"task":"/task/clinical-assertion-status-detection","name":"Clinical Assertion Status Detection","papers":1},{"task":"/task/clinical-concept-extraction","name":"Clinical Concept Extraction","papers":1},{"task":"/task/drug-discovery","name":"Drug Discovery","papers":1},{"task":"/task/feature-engineering","name":"Feature Engineering","papers":1},{"task":"/task/classification","name":"General Classification","papers":1},{"task":"/task/knowledge-graph-embeddings","name":"Knowledge Graph Embeddings","papers":1},{"task":"/task/language-modeling","name":"Language Modeling","papers":1},{"task":"/task/language-modelling","name":"Language Modelling","papers":1},{"task":"/task/link-prediction","name":"Link Prediction","papers":1},{"task":"/task/multiple-choice","name":"Multiple-choice","papers":1},{"task":"/task/named-entity-recognition-1","name":"Named Entity Recognition","papers":1},{"task":"/task/named-entity-recognition-ner","name":"Named Entity Recognition (NER)","papers":1},{"task":"/task/natural-language-inference","name":"Natural Language Inference","papers":1}],"tasks_shown":20,"n_tasks":38,"usage_by_year":[{"year":"2013","papers":1},{"year":"2017","papers":2},{"year":"2019","papers":2},{"year":"2020","papers":6},{"year":"2022","papers":2},{"year":"2023","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/skip-gram-word2vec"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}