{"url":"/dataset/tweebank","name":"Tweebank","full_name":null,"description_markdown":"Briefly describe the dataset. Provide:\r\n\r\n* a high-level explanation of the dataset characteristics\r\n* explain motivations and summary of its content\r\n* potential use cases of the dataset\r\n\r\nIf the description or image is from a different paper, please refer to it as follows:\r\nSource: [title](url)\r\nImage Source: [title](url)","description_withheld":null,"homepage":"https://github.com/Oneplus/Tweebank","introduced_date":"2018-04-23","introduced_date_note":null,"introduced_by":{"paper":"/paper/parsing-tweets-into-universal-dependencies","title":"Parsing Tweets into Universal Dependencies","first_author":"Yijia Liu","url":null},"license":null,"modalities":[],"tasks":[{"name":"Dependency Parsing","url":"/task/dependency-parsing","datasets_with_task":"/datasets/task/dependency-parsing"},{"name":"Part-Of-Speech Tagging","url":"/task/part-of-speech-tagging","datasets_with_task":"/datasets/task/part-of-speech-tagging"}],"languages":[],"variants":["Tweebank"],"data_loaders":[],"num_papers_in_archive":20,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/dependency-parsing-on-tweebank","task":"Dependency Parsing","dataset_variant":"Tweebank","rows":3,"metrics":["Labelled Attachment Score","Unlabeled Attachment Score"],"first_row_in_archive_order":{"model":"SuPar-BERTweet","paper":"/paper/cross-dialect-social-media-dependency-parsing","metrics":{"Labelled Attachment Score":"83.4","Unlabeled Attachment Score":"87.2 "},"code_links":[{"title":"slanglab/tweetie_wnut2022","url":"https://github.com/slanglab/tweetie_wnut2022"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/part-of-speech-tagging-on-tweebank","task":"Part-Of-Speech Tagging","dataset_variant":"Tweebank","rows":3,"metrics":["Acc"],"first_row_in_archive_order":{"model":"ACE","paper":"/paper/automated-concatenation-of-embeddings-for-1","metrics":{"Acc":"95.8"},"code_links":[{"title":"Alibaba-NLP/ACE","url":"https://github.com/Alibaba-NLP/ACE"},{"title":"zhaoyuesun/phee","url":"https://github.com/zhaoyuesun/phee"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/cross-dialect-social-media-dependency-parsing","title":"Cross-Dialect Social Media Dependency Parsing for Social Scientific Entity Attribute Analysis","date":"2022-10-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/annotating-the-tweebank-corpus-on-named","title":"Annotating the Tweebank Corpus on Named Entity Recognition and Building NLP Models for Social Media Analysis","date":"2022-01-18","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":5,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/automated-concatenation-of-embeddings-for-1","title":"Automated Concatenation of Embeddings for Structured Prediction","date":"2020-10-10","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/bertweet-a-pre-trained-language-model-for","title":"BERTweet: A pre-trained language model for English Tweets","date":"2020-05-20","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/parsing-tweets-into-universal-dependencies","title":"Parsing Tweets into Universal Dependencies","date":"2018-04-23","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/part-of-speech-tagging-for-twitter-with","title":"Part-of-Speech Tagging for Twitter with Adversarial Neural Networks","date":"2017-09-01","rows_on_this_dataset":1,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":15,"samples_ran":5,"samples_unverified":10,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}