{"url":"/dataset/risawoz","name":"RiSAWOZ","full_name":null,"description_markdown":"**RiSAWOZ** is a large-scale multi-domain Chinese Wizard-of-Oz dataset with Rich Semantic Annotations. RiSAWOZ contains 11.2K human-to-human (H2H) multi-turn semantically annotated dialogues, with more than 150K utterances spanning over 12 domains, which is larger than all previous annotated H2H conversational datasets. Both single- and multi-domain dialogues are constructed, accounting for 65% and 35%, respectively. Each dialogue is labelled with comprehensive dialogue annotations, including dialogue goal in the form of natural language description, domain, dialogue states and acts at both the user and system side. In addition to traditional dialogue annotations, it also includes linguistic annotations on discourse phenomena, e.g., ellipsis and coreference, in dialogues, which are useful for dialogue coreference and ellipsis resolution tasks.\n\nSource: [https://github.com/terryqj0107/RiSAWOZ](https://github.com/terryqj0107/RiSAWOZ)","description_withheld":null,"homepage":"https://github.com/terryqj0107/RiSAWOZ","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/risawoz-a-large-scale-multi-domain-wizard-of","title":"RiSAWOZ: A Large-Scale Multi-Domain Wizard-of-Oz Dataset with Rich Semantic Annotations for Task-Oriented Dialogue Modeling","first_author":"Jun Quan","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Text Generation","url":"/task/text-generation","datasets_with_task":"/datasets/task/text-generation"},{"name":"Slot Filling","url":"/task/slot-filling","datasets_with_task":"/datasets/task/slot-filling"},{"name":"Dialogue State Tracking","url":"/task/dialogue-state-tracking","datasets_with_task":"/datasets/task/dialogue-state-tracking"}],"languages":[{"name":"Chinese","url":"/datasets/language/chinese"}],"variants":["RiSAWOZ"],"data_loaders":[{"repo":"https://github.com/terryqj0107/RiSAWOZ","url":"https://github.com/terryqj0107/RiSAWOZ","frameworks":[]}],"num_papers_in_archive":5,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}