{"url":"/dataset/tweetqa","name":"TweetQA","full_name":null,"description_markdown":"With social media becoming increasingly popular on which lots of news and real-time events are reported, developing automated question answering systems is critical to the effectiveness of many applications that rely on real-time knowledge. While previous question answering (QA) datasets have concentrated on formal text like news and Wikipedia, the first large-scale dataset for QA over social media data is presented. To make sure the tweets are meaningful and contain interesting information, tweets used by journalists to write news articles are gathered. Then human annotators are asked to write questions and answers upon these tweets. Unlike other QA datasets like SQuAD in which the answers are extractive, the answer are allowed to be abstractive. The task requires model to read a short tweet and a question and outputs a text phrase (does not need to be in the tweet) as the answer.\r\n\r\nSource: [TWEETQA: A Social Media Focused Question Answering Dataset](https://tweetqa.github.io/)","description_withheld":null,"homepage":"https://tweetqa.github.io/","introduced_date":"2019-07-14","introduced_date_note":null,"introduced_by":{"paper":"/paper/tweetqa-a-social-media-focused-question","title":"TWEETQA: A Social Media Focused Question Answering Dataset","first_author":"Wenhan Xiong","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"Continual Learning","url":"/task/continual-learning","datasets_with_task":"/datasets/task/continual-learning"},{"name":"Machine Reading Comprehension","url":"/task/machine-reading-comprehension","datasets_with_task":"/datasets/task/machine-reading-comprehension"}],"languages":[],"variants":["TweetQA"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/ucsbnlp/tweet_qa","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/tweet_qa","frameworks":["tf","pytorch","jax"]}],"num_papers_in_archive":19,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/question-answering-on-tweetqa","task":"Question Answering","dataset_variant":"TweetQA","rows":3,"metrics":["BLEU-1","ROUGE-L"],"first_row_in_archive_order":{"model":"ByT5 (small)","paper":"/paper/byt5-towards-a-token-free-future-with-pre","metrics":{"BLEU-1":"72.0"},"code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers/tree/master/src/transformers/models/byt5"},{"title":"google-research/byt5","url":"https://github.com/google-research/byt5"},{"title":"ufal/multilexnorm2021","url":"https://github.com/ufal/multilexnorm2021"},{"title":"2024-MindSpore-1/Code2","url":"https://github.com/2024-MindSpore-1/Code2/tree/main/model-1/byt5"},{"title":"yoreG123/Paddle-ByT5","url":"https://github.com/yoreG123/Paddle-ByT5"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/byt5-towards-a-token-free-future-with-pre","title":"ByT5: Towards a token-free future with pre-trained byte-to-byte models","date":"2021-05-28","rows_on_this_dataset":3,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":0,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":6,"samples_ran":0,"samples_unverified":6,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}