{"url":"/dataset/cais","name":"CAIS","full_name":"Chinese Artificial Intelligence Speakers","description_markdown":"We collect utterances from the Chinese Artificial Intelligence Speakers (CAIS), and an\u0002notate them with slot tags and intent labels. The training, validation and test sets are split by the distribution of intents, where detailed statistics are provided in the supplementary material. Since the utterances are collected from speaker systems in the real world, intent labels are partial to the PlayMusic option. We adopt the BIOES tag\u0002ging scheme for slots instead of the BIO2 used in the ATIS, since previous studies have highlighted meaningful improvements with this scheme (Rati\u0002nov and Roth, 2009) in the sequence labeling field","description_withheld":null,"homepage":"https://github.com/Adaxry/CM-Net/tree/master/CAIS","introduced_date":"2019-09-16","introduced_date_note":null,"introduced_by":{"paper":"/paper/cm-net-a-novel-collaborative-memory-network","title":"CM-Net: A Novel Collaborative Memory Network for Spoken Language Understanding","first_author":"Yijin Liu","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Intent Detection","url":"/task/intent-detection","datasets_with_task":"/datasets/task/intent-detection"},{"name":"Slot Filling","url":"/task/slot-filling","datasets_with_task":"/datasets/task/slot-filling"}],"languages":[{"name":"Chinese","url":"/datasets/language/chinese"}],"variants":["CAIS"],"data_loaders":[],"num_papers_in_archive":6,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/intent-detection-on-cais","task":"Intent Detection","dataset_variant":"CAIS","rows":1,"metrics":["Acc"],"first_row_in_archive_order":{"model":"CM-Net","paper":"/paper/cm-net-a-novel-collaborative-memory-network","metrics":{"Acc":"94.56"},"code_links":[{"title":"Adaxry/CM-Net","url":"https://github.com/Adaxry/CM-Net"},{"title":"1053399472/CAISandSMP","url":"https://github.com/1053399472/CAISandSMP"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/slot-filling-on-cais","task":"Slot Filling","dataset_variant":"CAIS","rows":1,"metrics":["F1"],"first_row_in_archive_order":{"model":"CM-Net","paper":"/paper/cm-net-a-novel-collaborative-memory-network","metrics":{"F1":"86.16"},"code_links":[{"title":"Adaxry/CM-Net","url":"https://github.com/Adaxry/CM-Net"},{"title":"1053399472/CAISandSMP","url":"https://github.com/1053399472/CAISandSMP"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/cm-net-a-novel-collaborative-memory-network","title":"CM-Net: A Novel Collaborative Memory Network for Spoken Language Understanding","date":"2019-09-16","rows_on_this_dataset":2,"code_links":2,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}