{"url":"/task/keyword-extraction","name":"Keyword Extraction","slug":"keyword-extraction","description_markdown":"Keyword extraction is tasked with the automatic identification of terms that best describe the subject of a document (Source: Wikipedia).","categories":[{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":172,"papers_with_code":33,"benchmarks":3,"benchmark_tables_in_archive":3,"benchmark_tables_shown":3,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":7,"subtasks":0,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/keyword-extraction-on-inspec","slug":"keyword-extraction-on-inspec","dataset":"Inspec","dataset_url":"/dataset/inspec","rows_in_archive":5,"metrics":["F1 score","Precision@10","Recall @ 10"],"first_row_in_archive_order":{"model":"Phraseformer(BERT, ExEm(ft))","paper_title":"Phraseformer: Multimodal Key-phrase Extraction using Transformer and Graph Embedding","paper_url":"/paper/phraseformer-multimodal-key-phrase-extraction","paper_date":"2021-06-09","arxiv_id":"2106.04939","code_links":[],"syntology":null}},{"leaderboard":"/sota/keyword-extraction-on-semeval-2010-task-8","slug":"keyword-extraction-on-semeval-2010-task-8","dataset":"SemEval 2010 Task 8","dataset_url":null,"rows_in_archive":5,"metrics":["F1 score","Precision@10","Recall@10"],"first_row_in_archive_order":{"model":"Phraseformer(BERT, ExEm(ft))","paper_title":"Phraseformer: Multimodal Key-phrase Extraction using Transformer and Graph Embedding","paper_url":"/paper/phraseformer-multimodal-key-phrase-extraction","paper_date":"2021-06-09","arxiv_id":"2106.04939","code_links":[],"syntology":null}},{"leaderboard":"/sota/keyword-extraction-on-semeval2017","slug":"keyword-extraction-on-semeval2017","dataset":"SemEval-2017 Task-10","dataset_url":"/dataset/semeval2017","rows_in_archive":5,"metrics":["F1 score","Precision@10","Recall@10"],"first_row_in_archive_order":{"model":"Phraseformer(BERT, ExEm(ft))","paper_title":"Phraseformer: Multimodal Key-phrase Extraction using Transformer and Graph Embedding","paper_url":"/paper/phraseformer-multimodal-key-phrase-extraction","paper_date":"2021-06-09","arxiv_id":"2106.04939","code_links":[],"syntology":null}}],"datasets":[{"url":"/dataset/mpqa-opinion-corpus","name":"MPQA Opinion Corpus","full_name":"Multi-Perspective Question Answering","num_papers_in_archive":313},{"url":"/dataset/semeval2017","name":"SemEval-2017 Task-10","full_name":"","num_papers_in_archive":33},{"url":"/dataset/kptimes","name":"KPTimes","full_name":"","num_papers_in_archive":27},{"url":"/dataset/inspec","name":"Inspec","full_name":"","num_papers_in_archive":7},{"url":"/dataset/csl-2022","name":"CSL (Chinese Scientific Literature)","full_name":"","num_papers_in_archive":1},{"url":"/dataset/keyphrases-cs-math-russian","name":"Keyphrases CS&Math Russian","full_name":"Keyphrases CS&Math Russian","num_papers_in_archive":1},{"url":"/dataset/maked","name":"MAKED","full_name":"MultiModal MultiLingual Summarization and Keyword Extraction Dataset","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":33,"tagged_in_all":172,"items":[{"url":"/paper/181110831","title":"sCAKE: Semantic Connectivity Aware Keyword Extraction","date":"2018-11-27","arxiv_id":"1811.10831","repositories_listed":5,"syntology":null},{"url":"/paper/complex-network-based-supervised-keyword","title":"Complex Network based Supervised Keyword Extractor","date":"2019-09-26","arxiv_id":"1909.12009","repositories_listed":2,"syntology":null},{"url":"/paper/pscon-toward-conversational-product-search","title":"PSCon: Product Search Through Conversations","date":"2025-02-19","arxiv_id":"2502.13881","repositories_listed":1,"syntology":null},{"url":"/paper/seke-specialised-experts-for-keyword","title":"SEKE: Specialised Experts for Keyword Extraction","date":"2024-12-18","arxiv_id":"2412.14087","repositories_listed":1,"syntology":null},{"url":"/paper/llms-are-biased-evaluators-but-not-biased-for","title":"LLMs are Biased Evaluators But Not Biased for Retrieval Augmented Generation","date":"2024-10-28","arxiv_id":"2410.20833","repositories_listed":1,"syntology":null},{"url":"/paper/robust-privacy-amidst-innovation-with-large","title":"Robust Privacy Amidst Innovation with Large Language Models Through a Critical Assessment of the Risks","date":"2024-07-23","arxiv_id":"2407.16166","repositories_listed":1,"syntology":null},{"url":"/paper/an-improved-method-for-class-specific-keyword","title":"An Improved Method for Class-specific Keyword Extraction: A Case Study in the German Business Registry","date":"2024-07-19","arxiv_id":"2407.14085","repositories_listed":1,"syntology":null},{"url":"/paper/anytasktune-advanced-domain-specific","title":"AnyTaskTune: Advanced Domain-Specific Solutions through Task-Fine-Tuning","date":"2024-07-09","arxiv_id":"2407.07094","repositories_listed":1,"syntology":null},{"url":"/paper/open-world-multi-label-text-classification","title":"Open-world Multi-label Text Classification with Extremely Weak Supervision","date":"2024-07-08","arxiv_id":"2407.05609","repositories_listed":1,"syntology":null},{"url":"/paper/ksw-khmer-stop-word-based-dictionary-for","title":"KSW: Khmer Stop Word based Dictionary for Keyword Extraction","date":"2024-05-27","arxiv_id":"2405.17390","repositories_listed":1,"syntology":null},{"url":"/paper/speechclip-self-supervised-multi-task","title":"SpeechCLIP+: Self-supervised multi-task representation learning for speech via CLIP and speech-image data","date":"2024-02-10","arxiv_id":"2402.06959","repositories_listed":1,"syntology":null},{"url":"/paper/task-oriented-conversational-modelling-with","title":"Task Oriented Conversational Modelling With Subjective Knowledge","date":"2023-03-30","arxiv_id":"2303.17695","repositories_listed":1,"syntology":null},{"url":"/paper/graph-based-semantical-extractive-text","title":"Graph-based Semantical Extractive Text Analysis","date":"2022-12-19","arxiv_id":"2212.09701","repositories_listed":1,"syntology":null},{"url":"/paper/adaptkeybert-an-attention-based-approach","title":"AdaptKeyBERT: An Attention-Based approach towards Few-Shot & Zero-Shot Domain Adaptation of KeyBERT","date":"2022-11-14","arxiv_id":"2211.07499","repositories_listed":1,"syntology":null},{"url":"/paper/large-scale-bidirectional-training-for-zero","title":"Large-Scale Bidirectional Training for Zero-Shot Image Captioning","date":"2022-11-13","arxiv_id":"2211.06774","repositories_listed":1,"syntology":null},{"url":"/paper/sidi-kws-a-large-scale-multilingual-dataset","title":"SiDi KWS: A Large-Scale Multilingual Dataset for Keyword Spotting","date":"2022-09-22","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/study-of-keyword-extraction-techniques-for","title":"Study of keyword extraction techniques for Electric Double Layer Capacitor domain using text similarity indexes: An experimental analysis","date":"2021-11-13","arxiv_id":"2111.07068","repositories_listed":1,"syntology":null},{"url":"/paper/mderank-a-masked-document-embedding-rank","title":"MDERank: A Masked Document Embedding Rank Approach for Unsupervised Keyphrase Extraction","date":"2021-10-13","arxiv_id":"2110.06651","repositories_listed":1,"syntology":null},{"url":"/paper/drift-a-toolkit-for-diachronic-analysis-of","title":"DRIFT: A Toolkit for Diachronic Analysis of Scientific Literature","date":"2021-07-02","arxiv_id":"2107.01198","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/back-to-the-basics-a-quantitative-analysis-of","title":"Back to the Basics: A Quantitative Analysis of Statistical and Graph-Based Term Weighting Schemes for Keyword Extraction","date":"2021-04-16","arxiv_id":"2104.08028","repositories_listed":1,"syntology":{"n":6,"n_ran":0,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/frake-fusional-real-time-automatic-keyword","title":"FRAKE: Fusional Real-time Automatic Keyword Extraction","date":"2021-04-10","arxiv_id":"2104.04830","repositories_listed":1,"syntology":null},{"url":"/paper/elske-efficient-large-scale-keyphrase","title":"ELSKE: Efficient Large-Scale Keyphrase Extraction","date":"2021-02-10","arxiv_id":"2102.05700","repositories_listed":1,"syntology":null},{"url":"/paper/extending-neural-keyword-extraction-with-tf","title":"Extending Neural Keyword Extraction with TF-IDF tagset matching","date":"2021-01-31","arxiv_id":"2102.00472","repositories_listed":1,"syntology":null},{"url":"/paper/outline-to-story-fine-grained-controllable","title":"Outline to Story: Fine-grained Controllable Story Generation from Cascaded Events","date":"2021-01-04","arxiv_id":"2101.00822","repositories_listed":1,"syntology":null},{"url":"/paper/literature-retrieval-for-precision-medicine","title":"Literature Retrieval for Precision Medicine with Neural Matching and Faceted Summarization","date":"2020-12-17","arxiv_id":"2012.09355","repositories_listed":1,"syntology":null},{"url":"/paper/keywords-lie-far-from-the-mean-of-all-words","title":"Keywords lie far from the mean of all words in local vector space","date":"2020-08-21","arxiv_id":"2008.09513","repositories_listed":1,"syntology":null},{"url":"/paper/tnt-kid-transformer-based-neural-tagger-for","title":"TNT-KID: Transformer-based Neural Tagger for Keyword Identification","date":"2020-03-20","arxiv_id":"2003.09166","repositories_listed":1,"syntology":null},{"url":"/paper/semantic-sensitive-tf-idf-to-determine-word","title":"Semantic Sensitive TF-IDF to Determine Word Relevance in Documents","date":"2020-01-06","arxiv_id":"2001.09896","repositories_listed":1,"syntology":null},{"url":"/paper/rakun-rank-based-keyword-extraction-via","title":"RaKUn: Rank-based Keyword extraction via Unsupervised learning and Meta vertex aggregation","date":"2019-07-15","arxiv_id":"1907.06458","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-generation-and-processing-of-word","title":"Efficient Generation and Processing of Word Co-occurrence Networks Using corpus2graph","date":"2018-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null}],"syntology_records":2,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}