{"url":"/dataset/reddit-conversation-corpus","name":"Reddit Conversation Corpus","full_name":null,"description_markdown":"Reddit Conversation Corpus (RCC) consists of conversations, scraped from Reddit, for a 20 month period from November 2016 until August 2018. To ensure the quality and diversity of topics, 95 subreddits are selected from which conversations are collected. In total, RCC contains 9.2 million 3-turn conversations.","description_withheld":null,"homepage":"https://github.com/nouhadziri/THRED","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/augmenting-neural-response-generation-with","title":"Augmenting Neural Response Generation with Context-Aware Topical Attention","first_author":"Nouha Dziri","url":null},"license":{"name":"MIT","url":"https://github.com/nouhadziri/THRED/blob/master/LICENSE"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Open-Domain Dialog","url":"/task/open-domain-dialog","datasets_with_task":"/datasets/task/open-domain-dialog"}],"languages":[{"name":"Spanish","url":"/datasets/language/spanish"}],"variants":["Reddit Conversation Corpus"],"data_loaders":[],"num_papers_in_archive":6,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}