{"url":"/dataset/ubuntu-dialogue-corpus","name":"UDC","full_name":"Ubuntu Dialogue Corpus","description_markdown":"**Ubuntu Dialogue Corpus** (**UDC**) is a dataset containing almost 1 million multi-turn dialogues, with a total of over 7 million utterances and 100 million words. This provides a unique resource for research into building dialogue managers based on neural language models that can make use of large amounts of unlabeled data. The dataset has both the multi-turn property of conversations in the Dialog State Tracking Challenge datasets, and the unstructured nature of interactions from microblog services such as Twitter. \r\n\r\nSource: [The Ubuntu Dialogue Corpus: A Large Dataset for Research in Unstructured Multi-Turn Dialogue Systems](https://arxiv.org/pdf/1506.08909v3.pdf)","description_withheld":null,"homepage":"https://github.com/npow/ubottu","introduced_date":"2016-02-04","introduced_date_note":null,"introduced_by":{"paper":"/paper/the-ubuntu-dialogue-corpus-a-large-dataset-1","title":"The Ubuntu Dialogue Corpus: A Large Dataset for Research in Unstructured Multi-Turn Dialogue Systems","first_author":"Ryan Lowe","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"},{"name":"Dialog","url":"/datasets/modality/dialog"}],"tasks":[{"name":"Conversational Response Selection","url":"/task/conversational-response-selection","datasets_with_task":"/datasets/task/conversational-response-selection"},{"name":"Dialogue Generation","url":"/task/dialogue-generation","datasets_with_task":"/datasets/task/dialogue-generation"},{"name":"Answer Selection","url":"/task/answer-selection","datasets_with_task":"/datasets/task/answer-selection"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["Ubuntu Dialogue (Activity)","Ubuntu Dialogue (Entity)","Ubuntu Dialogue (Tense)","Ubuntu Dialogue (Cmd)","Ubuntu Dialogue (v1, Ranking)","Ubuntu Dialogue (v2, Ranking)","UDC"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/ubuntu-dialogs-corpus/ubuntu_dialogs_corpus","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/ubuntu_dialogs_corpus","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/facebookresearch/ParlAI","url":"https://parl.ai/docs/tasks.html#ubuntu","frameworks":["pytorch"]},{"repo":"https://github.com/npow/ubottu","url":"https://github.com/npow/ubottu","frameworks":[]}],"num_papers_in_archive":46,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/conversational-response-selection-on-ubuntu-1","task":"Conversational Response Selection","dataset_variant":"Ubuntu Dialogue (v1, Ranking)","rows":25,"metrics":["R10@1","R10@2","R10@5","R2@1"],"first_row_in_archive_order":{"model":"Dial-MAE","paper":"/paper/contextual-masked-auto-encoder-for-retrieval","metrics":{"R10@1":"0.918","R10@2":"0.964","R10@5":"0.993"},"code_links":[{"title":"suu990901/Dial-MAE","url":"https://github.com/suu990901/Dial-MAE"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/answer-selection-on-ubuntu-dialogue-v2","task":"Answer Selection","dataset_variant":"Ubuntu Dialogue (v2, Ranking)","rows":2,"metrics":["1 in 10 R@1","1 in 10 R@2","1 in 10 R@5","1 in 2 R@1"],"first_row_in_archive_order":{"model":"BERT + Keep Learning","paper":"/paper/keep-learning-self-supervised-meta-learning","metrics":{"1 in 10 R@1":"82.4"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/answer-selection-on-ubuntu-dialogue-v1","task":"Answer Selection","dataset_variant":"Ubuntu Dialogue (v1, Ranking)","rows":1,"metrics":["1 in 10 R@1","1 in 10 R@2","1 in 10 R@5","1 in 2 R@1"],"first_row_in_archive_order":{"model":"HRDE-LTC","paper":"/paper/learning-to-rank-question-answer-pairs-using","metrics":{"1 in 10 R@1":"0.684","1 in 10 R@2":"0.822","1 in 10 R@5":"0.960","1 in 2 R@1":"0.916"},"code_links":[{"title":"david-yoon/QA_HRDE_LTC","url":"https://github.com/david-yoon/QA_HRDE_LTC"},{"title":"younggns/comparative-abusive-lang","url":"https://github.com/younggns/comparative-abusive-lang"},{"title":"aus10powell/Automated-Health-Responses","url":"https://github.com/aus10powell/Automated-Health-Responses"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/conversational-response-selection-on-ubuntu-2","task":"Conversational Response Selection","dataset_variant":"Ubuntu Dialogue (v2, Ranking)","rows":1,"metrics":["R10@1","R10@2","R10@5"],"first_row_in_archive_order":{"model":"Uni-Encoder","paper":"/paper/global-selector-a-new-benchmark-dataset-and","metrics":{"R10@1":"0.859","R10@2":"0.938","R10@5":"0.990"},"code_links":[{"title":"dll-wu/uni-encoder","url":"https://github.com/dll-wu/uni-encoder"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/dialogue-generation-on-ubuntu-dialogue","task":"Dialogue Generation","dataset_variant":"Ubuntu Dialogue (Activity)","rows":1,"metrics":["F1","Precision","Recall"],"first_row_in_archive_order":{"model":"MrRNN Act.-Ent.","paper":"/paper/multiresolution-recurrent-neural-networks-an","metrics":{"F1":"11.43","Precision":"16.84","Recall":"9.72"},"code_links":[{"title":"julianser/hed-dlg-truncated","url":"https://github.com/julianser/hed-dlg-truncated"},{"title":"julianser/Ubuntu-Multiresolution-Tools","url":"https://github.com/julianser/Ubuntu-Multiresolution-Tools"},{"title":"WolfNiu/AdversarialDialogue","url":"https://github.com/WolfNiu/AdversarialDialogue"},{"title":"wayalhruhi/julianser","url":"https://github.com/wayalhruhi/julianser"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/dialogue-generation-on-ubuntu-dialogue-cmd","task":"Dialogue Generation","dataset_variant":"Ubuntu Dialogue (Cmd)","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"MrRNN Act.-Ent.","paper":"/paper/multiresolution-recurrent-neural-networks-an","metrics":{"Accuracy":"95.04%"},"code_links":[{"title":"julianser/hed-dlg-truncated","url":"https://github.com/julianser/hed-dlg-truncated"},{"title":"julianser/Ubuntu-Multiresolution-Tools","url":"https://github.com/julianser/Ubuntu-Multiresolution-Tools"},{"title":"WolfNiu/AdversarialDialogue","url":"https://github.com/WolfNiu/AdversarialDialogue"},{"title":"wayalhruhi/julianser","url":"https://github.com/wayalhruhi/julianser"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/dialogue-generation-on-ubuntu-dialogue-entity","task":"Dialogue Generation","dataset_variant":"Ubuntu Dialogue (Entity)","rows":1,"metrics":["F1","Precision","Recall"],"first_row_in_archive_order":{"model":"MrRNN Act.-Ent.","paper":"/paper/multiresolution-recurrent-neural-networks-an","metrics":{"F1":"3.72","Precision":"4.91","Recall":"3.36"},"code_links":[{"title":"julianser/hed-dlg-truncated","url":"https://github.com/julianser/hed-dlg-truncated"},{"title":"julianser/Ubuntu-Multiresolution-Tools","url":"https://github.com/julianser/Ubuntu-Multiresolution-Tools"},{"title":"WolfNiu/AdversarialDialogue","url":"https://github.com/WolfNiu/AdversarialDialogue"},{"title":"wayalhruhi/julianser","url":"https://github.com/wayalhruhi/julianser"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/dialogue-generation-on-ubuntu-dialogue-tense","task":"Dialogue Generation","dataset_variant":"Ubuntu Dialogue (Tense)","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"MrRNN Act.-Ent.","paper":"/paper/multiresolution-recurrent-neural-networks-an","metrics":{"Accuracy":"29.01%"},"code_links":[{"title":"julianser/hed-dlg-truncated","url":"https://github.com/julianser/hed-dlg-truncated"},{"title":"julianser/Ubuntu-Multiresolution-Tools","url":"https://github.com/julianser/Ubuntu-Multiresolution-Tools"},{"title":"WolfNiu/AdversarialDialogue","url":"https://github.com/WolfNiu/AdversarialDialogue"},{"title":"wayalhruhi/julianser","url":"https://github.com/wayalhruhi/julianser"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/efficient-dynamic-hard-negative-sampling-for","title":"Efficient Dynamic Hard Negative Sampling for Dialogue Selection","date":"2024-08-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/contextual-masked-auto-encoder-for-retrieval","title":"Dial-MAE: ConTextual Masked Auto-Encoder for Retrieval-based Dialogue Systems","date":"2023-06-07","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/small-changes-make-big-differences-improving","title":"Small Changes Make Big Differences: Improving Multi-turn Response Selection in Dialogue Systems via Fine-Grained Contrastive Learning","date":"2021-11-19","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/response-ranking-with-multi-types-of-deep","title":"Response Ranking with Multi-types of Deep Interactive Representations in Retrieval-based Dialogues","date":"2021-08-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/global-selector-a-new-benchmark-dataset-and","title":"Uni-Encoder: A Fast and Accurate Response Selection Paradigm for Generation-Based Dialogue Systems","date":"2021-06-02","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/fine-grained-post-training-for-improving","title":"Fine-grained Post-training for Improving Retrieval-based Dialogue Systems","date":"2021-05-24","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/keep-learning-self-supervised-meta-learning","title":"Keep Learning: Self-supervised Meta-learning for Learning from Inference","date":"2021-04-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/learning-an-effective-context-response","title":"Learning an Effective Context-Response Matching Model with Self-Supervised Tasks for Retrieval-based Dialogues","date":"2020-09-14","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/do-response-selection-models-really-know-what","title":"Do Response Selection Models Really Know What's Next? Utterance Manipulation Strategies for Multi-turn Response Selection","date":"2020-09-10","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/speaker-aware-bert-for-multi-turn-response","title":"Speaker-Aware BERT for Multi-Turn Response Selection in Retrieval-Based Chatbots","date":"2020-04-07","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/sampling-matters-an-empirical-study-of","title":"Sampling Matters! An Empirical Study of Negative Sampling Strategies for Learning of Matching Models in Retrieval-based Dialogue Systems","date":"2019-11-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/multi-hop-selector-network-for-multi-turn","title":"Multi-hop Selector Network for Multi-turn Response Selection in Retrieval-based Chatbots","date":"2019-11-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/triplenet-triple-attention-network-for-multi","title":"TripleNet: Triple Attention Network for Multi-Turn Response Selection in Retrieval-based Chatbots","date":"2019-09-24","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/multi-granularity-representations-of-dialog","title":"Multi-Granularity Representations of Dialog","date":"2019-08-26","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/domain-adaptive-training-bert-for-response","title":"An Effective Domain Adaptive Post-Training Method for BERT in Response Selection","date":"2019-08-13","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/one-time-of-interaction-may-not-be-enough-go","title":"One Time of Interaction May Not Be Enough: Go Deep with an Interaction-over-Interaction Network for Response Selection in Dialogues","date":"2019-07-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/190501969","title":"Poly-encoders: Transformer Architectures and Pre-training Strategies for Fast and Accurate Multi-sentence Scoring","date":"2019-04-22","rows_on_this_dataset":1,"code_links":7,"syntology":null},{"paper":"/paper/sequential-attention-based-network-for-noetic","title":"Sequential Attention-based Network for Noetic End-to-End Response Selection","date":"2019-01-09","rows_on_this_dataset":1,"code_links":4,"syntology":null},{"paper":"/paper/interactive-matching-network-for-multi-turn","title":"Interactive Matching Network for Multi-Turn Response Selection in Retrieval-Based Chatbots","date":"2019-01-07","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/multi-turn-response-selection-for-chatbots","title":"Multi-Turn Response Selection for Chatbots with Deep Attention Matching Network","date":"2018-07-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/modeling-multi-turn-conversation-with-deep","title":"Modeling Multi-turn Conversation with Deep Utterance Aggregation","date":"2018-06-24","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/learning-to-rank-question-answer-pairs-using","title":"Learning to Rank Question-Answer Pairs using Hierarchical Recurrent Encoder with Latent Topic Clustering","date":"2017-10-10","rows_on_this_dataset":2,"code_links":3,"syntology":null},{"paper":"/paper/sequential-matching-network-a-new","title":"Sequential Matching Network: A New Architecture for Multi-turn Response Selection in Retrieval-based Chatbots","date":"2016-12-06","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":0,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multi-view-response-selection-for-human","title":"Multi-view Response Selection for Human-Computer Conversation","date":"2016-11-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/multiresolution-recurrent-neural-networks-an","title":"Multiresolution Recurrent Neural Networks: An Application to Dialogue Response Generation","date":"2016-06-02","rows_on_this_dataset":4,"code_links":4,"syntology":null},{"paper":"/paper/improved-deep-learning-baselines-for-ubuntu","title":"Improved Deep Learning Baselines for Ubuntu Corpus Dialogs","date":"2015-10-13","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/the-ubuntu-dialogue-corpus-a-large-dataset-1","title":"The Ubuntu Dialogue Corpus: A Large Dataset for Research in Unstructured Multi-Turn Dialogue Systems","date":"2015-06-30","rows_on_this_dataset":1,"code_links":21,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":26,"samples_ran":1,"samples_unverified":25,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":2,"samples_harvested":29,"samples_ran":1,"samples_unverified":28,"pointer_only_for_licence":1,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}