{"url":"/dataset/conll","name":"CoNLL++","full_name":null,"description_markdown":"CoNLL++ is a corrected version of the CoNLL03 NER dataset where 5.38% of the test sentences have been fixed.\r\n\r\nSource: [CrossWeigh: Training Named Entity Tagger from Imperfect Annotations](/paper/crossweigh-training-named-entity-tagger-from)","description_withheld":null,"homepage":"https://github.com/ZihanWangKi/CrossWeigh","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/crossweigh-training-named-entity-tagger-from","title":"CrossWeigh: Training Named Entity Tagger from Imperfect Annotations","first_author":"Zihan Wang","url":null},"license":null,"modalities":[],"tasks":[{"name":"Named Entity Recognition (NER)","url":"/task/named-entity-recognition-ner","datasets_with_task":"/datasets/task/named-entity-recognition-ner"},{"name":"Weakly-Supervised Named Entity Recognition","url":"/task/weakly-supervised-named-entity-recognition","datasets_with_task":"/datasets/task/weakly-supervised-named-entity-recognition"}],"languages":[],"variants":["CoNLL03","CoNLL++","conllpp"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/tomaarsen/conllpp","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/ZihanWangKi/conllpp","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/ncats/EpiSet4NER","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/conllpp","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/schaffen/conllpp","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/GregoireTR/output1","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/ZihanWangKi/CrossWeigh","url":"https://github.com/ZihanWangKi/CrossWeigh","frameworks":[]}],"num_papers_in_archive":56,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/named-entity-recognition-on-conll","task":"Named Entity Recognition (NER)","dataset_variant":"CoNLL++","rows":11,"metrics":["F1"],"first_row_in_archive_order":{"model":"LUKE + SubRegWeigh (K-means)","paper":"/paper/subregweigh-effective-and-efficient","metrics":{"F1":"96.12"},"code_links":[{"title":"4ldk/SubRegWeigh","url":"https://github.com/4ldk/SubRegWeigh"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/named-entity-recognition-on-conll03","task":"Named Entity Recognition (NER)","dataset_variant":"CoNLL03","rows":5,"metrics":["F1","F1 (micro)"],"first_row_in_archive_order":{"model":"UniNER-7B","paper":"/paper/universalner-targeted-distillation-from-large","metrics":{"F1":"93.3"},"code_links":[{"title":"universal-ner/universal-ner","url":"https://github.com/universal-ner/universal-ner"},{"title":"emma1066/retrieval-augmented-it-openner","url":"https://github.com/emma1066/retrieval-augmented-it-openner"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/subregweigh-effective-and-efficient","title":"SubRegWeigh: Effective and Efficient Annotation Weighing with Subword Regularization","date":"2024-09-10","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/label-supervised-llama-finetuning","title":"Label Supervised LLaMA Finetuning","date":"2023-10-02","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/universalner-targeted-distillation-from-large","title":"UniversalNER: Targeted Distillation from Large Language Models for Open Named Entity Recognition","date":"2023-08-07","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":5,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/deepstruct-pretraining-of-language-models-for-1","title":"DeepStruct: Pretraining of Language Models for Structure Prediction","date":"2022-05-21","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":7,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/improving-named-entity-recognition-by","title":"Improving Named Entity Recognition by External Context Retrieving and Cooperative Learning","date":"2021-05-08","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-from-noisy-labels-for-entity-centric","title":"Learning from Noisy Labels for Entity-Centric Information Extraction","date":"2021-04-17","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/luke-deep-contextualized-entity","title":"LUKE: Deep Contextualized Entity Representations with Entity-aware Self-attention","date":"2020-10-02","rows_on_this_dataset":1,"code_links":9,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":3,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/crossweigh-training-named-entity-tagger-from","title":"CrossWeigh: Training Named Entity Tagger from Imperfect Annotations","date":"2019-09-03","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":4,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/deep-contextualized-word-representations","title":"Deep contextualized word representations","date":"2018-02-15","rows_on_this_dataset":1,"code_links":46,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":58,"samples_ran":23,"samples_unverified":35,"pointer_only_for_licence":25,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/neural-architectures-for-named-entity","title":"Neural Architectures for Named Entity Recognition","date":"2016-03-04","rows_on_this_dataset":1,"code_links":43,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":36,"samples_ran":10,"samples_unverified":26,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/end-to-end-sequence-labeling-via-bi","title":"End-to-end Sequence Labeling via Bi-directional LSTM-CNNs-CRF","date":"2016-03-04","rows_on_this_dataset":1,"code_links":25,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":24,"samples_ran":4,"samples_unverified":20,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":8,"samples_harvested":161,"samples_ran":56,"samples_unverified":105,"pointer_only_for_licence":34,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}