{"url":"/dataset/broad-twitter-corpus","name":"Broad Twitter Corpus","full_name":null,"description_markdown":"This paper introduces the Broad Twitter Corpus (BTC), which is not only significantly bigger, but sampled across different regions, temporal periods, and types of Twitter users. The gold-standard named entity annotations are made by a combination of NLP experts and crowd workers, which enables us to harness crowd recall while maintaining high quality. We also measure the entity drift observed in our dataset (i.e. how entity representation varies over time), and compare to newswire.","description_withheld":null,"homepage":"https://github.com/GateNLP/broad_twitter_corpus","introduced_date":"2016-12-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/broad-twitter-corpus-a-diverse-named-entity","title":"Broad Twitter Corpus: A Diverse Named Entity Recognition Resource","first_author":"Leon Derczynski","url":null},"license":{"name":"CC-BY","url":null},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Named Entity Recognition (NER)","url":"/task/named-entity-recognition-ner","datasets_with_task":"/datasets/task/named-entity-recognition-ner"},{"name":"Zero-shot Named Entity Recognition (NER)","url":"/task/zero-shot-named-entity-recognition-ner","datasets_with_task":"/datasets/task/zero-shot-named-entity-recognition-ner"},{"name":"Low Resource Named Entity Recognition","url":"/task/low-resource-named-entity-recognition","datasets_with_task":"/datasets/task/low-resource-named-entity-recognition"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["Broad Twitter Corpus"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/GateNLP/broad_twitter_corpus","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/strombergnlp/broad_twitter_corpus","frameworks":["tf","pytorch","jax"]}],"num_papers_in_archive":12,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/zero-shot-named-entity-recognition-ner-on","task":"Zero-shot Named Entity Recognition (NER)","dataset_variant":"Broad Twitter Corpus","rows":2,"metrics":["Entity F1"],"first_row_in_archive_order":{"model":"NuNerZero Span","paper":"/paper/nuner-entity-recognition-encoder-pre-training","metrics":{"Entity F1":"60.2"},"code_links":[{"title":"Serega6678/NuNER","url":"https://github.com/Serega6678/NuNER"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/named-entity-recognition-on-broad-twitter","task":"Named Entity Recognition (NER)","dataset_variant":"Broad Twitter Corpus","rows":1,"metrics":["Entity F1"],"first_row_in_archive_order":{"model":"WORD_GAZ","paper":"/paper/the-utility-and-interplay-of-gazetteers-and","metrics":{"Entity F1":"74.70"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/nuner-entity-recognition-encoder-pre-training","title":"NuNER: Entity Recognition Encoder Pre-training via LLM-Annotated Data","date":"2024-02-23","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/gollie-annotation-guidelines-improve-zero","title":"GoLLIE: Annotation Guidelines improve Zero-Shot Information-Extraction","date":"2023-10-05","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":25,"samples_ran":18,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/the-utility-and-interplay-of-gazetteers-and","title":"The Utility and Interplay of Gazetteers and Entity Segmentation for Named Entity Recognition in English","date":"2021-08-01","rows_on_this_dataset":1,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":25,"samples_ran":18,"samples_unverified":7,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}