{"url":"/dataset/wnut-2017-emerging-and-rare-entity","name":"WNUT 2017","full_name":"WNUT 2017 Emerging and Rare entity recognition","description_markdown":"This shared task focuses on identifying unusual, previously-unseen entities in the context of emerging discussions. Named entities form the basis of many modern approaches to other tasks (like event clustering and summarisation), but recall on them is a real problem in noisy text - even among annotators. This drop tends to be due to novel entities and surface forms. Take for example the tweet “so.. kktny in 30 mins?” - even human experts find entity kktny hard to detect and resolve. This task will evaluate the ability to detect and classify novel, emerging, singleton named entities in noisy text.\r\n\r\nThe goal of this task is to provide a definition of emerging and of rare entities, and based on that, also datasets for detecting these entities.","description_withheld":null,"homepage":"https://noisy-text.github.io/2017/emerging-rare-entities.html","introduced_date":"2017-09-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/results-of-the-wnut2017-shared-task-on-novel","title":"Results of the WNUT2017 Shared Task on Novel and Emerging Entity Recognition","first_author":"Leon Derczynski","url":null},"license":{"name":"CC-BY 4.0","url":null},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Named Entity Recognition (NER)","url":"/task/named-entity-recognition-ner","datasets_with_task":"/datasets/task/named-entity-recognition-ner"},{"name":"UIE","url":"/task/uie","datasets_with_task":"/datasets/task/uie"},{"name":"Few-shot NER","url":"/task/few-shot-ner","datasets_with_task":"/datasets/task/few-shot-ner"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["Long-tail emerging entities","WNUT 2017","wnut_17"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/leondz/wnut_17","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/Jnanesh12/Slang","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/Kriyans/ner","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/Kriyans/indian_names","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/wnut_17","frameworks":["tf","pytorch","jax"]}],"num_papers_in_archive":127,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/named-entity-recognition-on-wnut-2017","task":"Named Entity Recognition (NER)","dataset_variant":"WNUT 2017","rows":23,"metrics":["F1","F1 (surface form)","Precision","Recall"],"first_row_in_archive_order":{"model":"CL-KL","paper":"/paper/improving-named-entity-recognition-by","metrics":{"F1":"60.45"},"code_links":[{"title":"modelscope/adaseq","url":"https://github.com/modelscope/adaseq"},{"title":"modelscope/AdaSeq","url":"https://github.com/modelscope/AdaSeq/tree/master/examples/RaNER"},{"title":"Alibaba-NLP/CLNER","url":"https://github.com/Alibaba-NLP/CLNER"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/uie-on-wnut-2017","task":"UIE","dataset_variant":"WNUT 2017","rows":1,"metrics":["F1 score"],"first_row_in_archive_order":{"model":"KnowCoder-7b-IE","paper":"/paper/knowcoder-coding-structured-knowledge-into","metrics":{"F1 score":"66.4"},"code_links":[{"title":"ICT-GoKnow/KnowCoder","url":"https://github.com/ICT-GoKnow/KnowCoder"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/subregweigh-effective-and-efficient","title":"SubRegWeigh: Effective and Efficient Annotation Weighing with Subword Regularization","date":"2024-09-10","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/knowcoder-coding-structured-knowledge-into","title":"KnowCoder: Coding Structured Knowledge into LLMs for Universal Information Extraction","date":"2024-03-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/gollie-annotation-guidelines-improve-zero","title":"GoLLIE: Annotation Guidelines improve Zero-Shot Information-Extraction","date":"2023-10-05","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":25,"samples_ran":18,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/a-global-context-mechanism-for-sequence","title":"Supplementary Features of BiLSTM for Enhanced Sequence Labeling","date":"2023-05-31","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/t-ner-an-all-round-python-library-for-1","title":"T-NER: An All-Round Python Library for Transformer-based Named Entity Recognition","date":"2022-09-09","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/adversarial-self-attention-for-language","title":"Adversarial Self-Attention for Language Understanding","date":"2022-06-25","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/hero-gang-neural-model-for-named-entity","title":"Hero-Gang Neural Model For Named Entity Recognition","date":"2022-05-15","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/miner-improving-out-of-vocabulary-named-1","title":"MINER: Improving Out-of-Vocabulary Named Entity Recognition from an Information Theoretic Perspective","date":"2022-04-09","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/improving-named-entity-recognition-by","title":"Improving Named Entity Recognition by External Context Retrieving and Cooperative Learning","date":"2021-05-08","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/inferner-an-attentive-model-leveraging-the","title":"InferNER: an attentive model leveraging the sentence-level information for Named Entity Recognition in Microblogs","date":"2021-04-18","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/regularizing-models-via-pointwise-mutual","title":"Regularization for Long Named Entity Recognition","date":"2021-04-15","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/named-entity-recognition-for-social-media","title":"Named Entity Recognition for Social Media Texts with Semantic Augmentation","date":"2020-10-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/improving-named-entity-recognition-with","title":"Improving Named Entity Recognition with Attentive Ensemble of Syntactic Information","date":"2020-10-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/bertweet-a-pre-trained-language-model-for","title":"BERTweet: A pre-trained language model for English Tweets","date":"2020-05-20","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/robust-named-entity-recognition-with","title":"Robust Named Entity Recognition with Truecasing Pretraining","date":"2019-12-15","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/similarity-based-auxiliary-classifier-for","title":"Similarity Based Auxiliary Classifier for Named Entity Recognition","date":"2019-11-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/crossweigh-training-named-entity-tagger-from","title":"CrossWeigh: Training Named Entity Tagger from Imperfect Annotations","date":"2019-09-03","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":4,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/remedying-bilstm-cnn-deficiency-in-modeling","title":"Why Attention? Analyze BiLSTM Deficiency and Its Remedies in the Case of NER","date":"2019-08-29","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/modeling-noisiness-to-recognize-named-1","title":"Modeling Noisiness to Recognize Named Entities using Multitask Neural Networks on Social Media","date":"2019-06-10","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/a-multi-task-approach-for-named-entity-1","title":"A Multi-task Approach for Named Entity Recognition in Social Media Data","date":"2019-06-10","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/transfer-learning-and-sentence-level-features","title":"Transfer Learning and Sentence Level Features for Named Entity Recognition on Tweets","date":"2017-09-01","rows_on_this_dataset":1,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":5,"samples_harvested":38,"samples_ran":25,"samples_unverified":13,"pointer_only_for_licence":3,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}