{"url":"/task/named-entity-recognition-1","name":"Named Entity Recognition","slug":"named-entity-recognition-1","description_markdown":null,"categories":[],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":2560,"papers_with_code":969,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":20,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[{"url":"/dataset/conll-2003","name":"CoNLL 2003","full_name":"","num_papers_in_archive":755},{"url":"/dataset/ontonotes-5-0","name":"OntoNotes 5.0","full_name":"","num_papers_in_archive":254},{"url":"/dataset/conll-1","name":"CoNLL","full_name":"","num_papers_in_archive":187},{"url":"/dataset/ncbi-disease-1","name":"NCBI Disease","full_name":"","num_papers_in_archive":154},{"url":"/dataset/scierc","name":"SciERC","full_name":"","num_papers_in_archive":134},{"url":"/dataset/conll-2012-1","name":"CoNLL-2012","full_name":"","num_papers_in_archive":89},{"url":"/dataset/few-nerd","name":"Few-NERD","full_name":"Few-NERD","num_papers_in_archive":77},{"url":"/dataset/wikiann-1","name":"WikiANN","full_name":"PAN-X","num_papers_in_archive":72},{"url":"/dataset/masakhaner","name":"MasakhaNER","full_name":"","num_papers_in_archive":56},{"url":"/dataset/gum","name":"GUM","full_name":"Georgetown University Multilayer corpus","num_papers_in_archive":13},{"url":"/dataset/nuner","name":"NuNER","full_name":"","num_papers_in_archive":4},{"url":"/dataset/latamxix","name":"LatamXIX","full_name":"19th Century Latin American Spanish Newspaper Corpus with LLM OCR Correction","num_papers_in_archive":2},{"url":"/dataset/coneco","name":"CoNECo","full_name":"Complex Named Entity Corpus","num_papers_in_archive":1},{"url":"/dataset/financial-dynamic-knowledge-graph","name":"Financial Dynamic Knowledge Graph","full_name":"","num_papers_in_archive":1},{"url":"/dataset/first-harem","name":"First HAREM","full_name":"Primeiro HAREM","num_papers_in_archive":1},{"url":"/dataset/mini-harem","name":"Mini HAREM","full_name":"","num_papers_in_archive":1},{"url":"/dataset/arf","name":"ARF","full_name":"Artificial Relationships in Fiction","num_papers_in_archive":0},{"url":"/dataset/popcorn","name":"POPCORN","full_name":"POPCORN: Fictional and Synthetic Intelligence Reports for Named Entity Recognition and Relation Extraction Tasks","num_papers_in_archive":0},{"url":"/dataset/second-harem","name":"Second HAREM","full_name":"Segundo HAREM","num_papers_in_archive":0},{"url":"/dataset/sigarra-news-corpus","name":"SIGARRA News Corpus","full_name":"","num_papers_in_archive":0}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":969,"tagged_in_all":2560,"items":[{"url":"/paper/bert-pre-training-of-deep-bidirectional","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","date":"2018-10-11","arxiv_id":"1810.04805","repositories_listed":534,"syntology":{"n":659,"n_ran":204,"n_unverified":455,"n_pointer_only":149}},{"url":"/paper/neural-architectures-for-named-entity","title":"Neural Architectures for Named Entity Recognition","date":"2016-03-04","arxiv_id":"1603.01360","repositories_listed":43,"syntology":{"n":36,"n_ran":10,"n_unverified":26,"n_pointer_only":6}},{"url":"/paper/end-to-end-sequence-labeling-via-bi","title":"End-to-end Sequence Labeling via Bi-directional LSTM-CNNs-CRF","date":"2016-03-04","arxiv_id":"1603.01354","repositories_listed":25,"syntology":{"n":24,"n_ran":4,"n_unverified":20,"n_pointer_only":3}},{"url":"/paper/ernie-enhanced-representation-through","title":"ERNIE: Enhanced Representation through Knowledge Integration","date":"2019-04-19","arxiv_id":"1904.09223","repositories_listed":19,"syntology":{"n":7,"n_ran":0,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/biobert-a-pre-trained-biomedical-language","title":"BioBERT: a pre-trained biomedical language representation model for biomedical text mining","date":"2019-01-25","arxiv_id":"1901.08746","repositories_listed":19,"syntology":{"n":25,"n_ran":4,"n_unverified":21,"n_pointer_only":1}},{"url":"/paper/named-entity-recognition-with-bidirectional","title":"Named Entity Recognition with Bidirectional LSTM-CNNs","date":"2015-11-26","arxiv_id":"1511.08308","repositories_listed":14,"syntology":{"n":9,"n_ran":4,"n_unverified":5,"n_pointer_only":4}},{"url":"/paper/nezha-neural-contextualized-representation","title":"NEZHA: Neural Contextualized Representation for Chinese Language Understanding","date":"2019-08-31","arxiv_id":"1909.00204","repositories_listed":10,"syntology":null},{"url":"/paper/luke-deep-contextualized-entity","title":"LUKE: Deep Contextualized Entity Representations with Entity-aware Self-attention","date":"2020-10-02","arxiv_id":"2010.01057","repositories_listed":9,"syntology":{"n":10,"n_ran":3,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/camembert-a-tasty-french-language-model","title":"CamemBERT: a Tasty French Language Model","date":"2019-11-10","arxiv_id":"1911.03894","repositories_listed":8,"syntology":null},{"url":"/paper/a-unified-mrc-framework-for-named-entity","title":"A Unified MRC Framework for Named Entity Recognition","date":"2019-10-25","arxiv_id":"1910.11476","repositories_listed":8,"syntology":{"n":5,"n_ran":2,"n_unverified":3,"n_pointer_only":2}},{"url":"/paper/few-nerd-a-few-shot-named-entity-recognition","title":"Few-NERD: A Few-Shot Named Entity Recognition Dataset","date":"2021-05-16","arxiv_id":"2105.07464","repositories_listed":7,"syntology":{"n":7,"n_ran":1,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/tener-adapting-transformer-encoder-for-name","title":"TENER: Adapting Transformer Encoder for Named Entity Recognition","date":"2019-11-10","arxiv_id":"1911.04474","repositories_listed":6,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/the-natural-language-decathlon-multitask","title":"The Natural Language Decathlon: Multitask Learning as Question Answering","date":"2018-06-20","arxiv_id":"1806.08730","repositories_listed":6,"syntology":{"n":13,"n_ran":1,"n_unverified":12,"n_pointer_only":4}},{"url":"/paper/xiyan-sql-a-multi-generator-ensemble","title":"A Preview of XiYan-SQL: A Multi-Generator Ensemble Framework for Text-to-SQL","date":"2024-11-13","arxiv_id":"2411.08599","repositories_listed":5,"syntology":null},{"url":"/paper/crossner-evaluating-cross-domain-named-entity","title":"CrossNER: Evaluating Cross-Domain Named Entity Recognition","date":"2020-12-08","arxiv_id":"2012.04373","repositories_listed":5,"syntology":null},{"url":"/paper/biomedical-and-clinical-english-model","title":"Biomedical and Clinical English Model Packages in the Stanza Python NLP Library","date":"2020-07-29","arxiv_id":"2007.14640","repositories_listed":5,"syntology":null},{"url":"/paper/stanza-a-python-natural-language-processing","title":"Stanza: A Python Natural Language Processing Toolkit for Many Human Languages","date":"2020-03-16","arxiv_id":"2003.07082","repositories_listed":5,"syntology":{"n":30,"n_ran":16,"n_unverified":14,"n_pointer_only":28}},{"url":"/paper/klue-korean-language-understanding-evaluation","title":"KLUE: Korean Language Understanding Evaluation","date":"2021-05-20","arxiv_id":"2105.09680","repositories_listed":4,"syntology":null},{"url":"/paper/neural-modeling-for-named-entities-and","title":"Neural Modeling for Named Entities and Morphology (NEMO^2)","date":"2020-07-30","arxiv_id":"2007.15620","repositories_listed":4,"syntology":null},{"url":"/paper/arabert-transformer-based-model-for-arabic","title":"AraBERT: Transformer-based Model for Arabic Language Understanding","date":"2020-02-28","arxiv_id":"2003.00104","repositories_listed":4,"syntology":null},{"url":"/paper/dice-loss-for-data-imbalanced-nlp-tasks","title":"Dice Loss for Data-imbalanced NLP Tasks","date":"2019-11-07","arxiv_id":"1911.02855","repositories_listed":4,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/entity-relation-and-event-extraction-with","title":"Entity, Relation, and Event Extraction with Contextualized Span Representations","date":"2019-09-08","arxiv_id":"1909.03546","repositories_listed":4,"syntology":{"n":7,"n_ran":0,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/cutie-learning-to-understand-documents-with","title":"CUTIE: Learning to Understand Documents with Convolutional Universal Text Information Extractor","date":"2019-03-29","arxiv_id":"1903.12363","repositories_listed":4,"syntology":null},{"url":"/paper/few-shot-learning-for-named-entity","title":"Few-shot Learning for Named Entity Recognition in Medical Text","date":"2018-11-13","arxiv_id":"1811.05468","repositories_listed":4,"syntology":null},{"url":"/paper/an-incremental-parser-for-abstract-meaning","title":"An Incremental Parser for Abstract Meaning Representation","date":"2016-08-22","arxiv_id":"1608.06111","repositories_listed":4,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/introduction-to-the-conll-2003-shared-task","title":"Introduction to the CoNLL-2003 Shared Task: Language-Independent Named Entity Recognition","date":"2003-06-12","arxiv_id":"cs/0306050","repositories_listed":4,"syntology":null},{"url":"/paper/rita-automatic-framework-for-designing-of","title":"RITA: Automatic Framework for Designing of Resilient IoT Applications","date":"2024-11-27","arxiv_id":"2411.18324","repositories_listed":3,"syntology":null},{"url":"/paper/diffusionner-boundary-diffusion-for-named","title":"DiffusionNER: Boundary Diffusion for Named Entity Recognition","date":"2023-05-22","arxiv_id":"2305.13298","repositories_listed":3,"syntology":null},{"url":"/paper/finer-financial-named-entity-recognition","title":"FiNER-ORD: Financial Named Entity Recognition Open Research Dataset","date":"2023-02-22","arxiv_id":"2302.11157","repositories_listed":3,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/xlm-v-overcoming-the-vocabulary-bottleneck-in","title":"XLM-V: Overcoming the Vocabulary Bottleneck in Multilingual Masked Language Models","date":"2023-01-25","arxiv_id":"2301.10472","repositories_listed":3,"syntology":null}],"syntology_records":16,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}