{"url":"/dataset/asr-ramc-bigccsc-a-chinese-conversational","name":"ASR-RAMC-BIGCCSC: A CHINESE CONVERSATIONAL SPEECH CORPUS","full_name":null,"description_markdown":"A Rich Annotated Mandarin Conversational (RAMC) Speech Dataset, including 180 hours of Mandarin Chinese dialogue, 150, 10 and 20 hours for the training set, development set and test set respectively.\r\nIt contains 351 multi-turn dialogues, each of which is a coherent and compact conversation centered around one theme.\r\n\r\nIt covers 15 topics, including humanities, entertainment, sports, military, finance, religion, family life, politics, education, digital devices, environment, science, professional development, art and ordinary life.\r\n\r\nIt is suitable for exploring speech processing techniques in dialog scenarios.","description_withheld":null,"homepage":"https://magichub.com/datasets/magicdata-ramc/","introduced_date":"2022-03-31","introduced_date_note":null,"introduced_by":{"paper":"/paper/open-source-magicdata-ramc-a-rich-annotated","title":"Open Source MagicData-RAMC: A Rich Annotated Mandarin Conversational(RAMC) Speech Dataset","first_author":"Zehui Yang","url":null},"license":{"name":"MAGIC DATA OPEN-SOURCE LICENSE","url":"https://magichub.com/magic-data-open-source-license/"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"},{"name":"Audio","url":"/datasets/modality/audio"}],"tasks":[{"name":"Speaker Diarization","url":"/task/speaker-diarization","datasets_with_task":"/datasets/task/speaker-diarization"},{"name":"Speaker Recognition","url":"/task/speaker-recognition","datasets_with_task":"/datasets/task/speaker-recognition"}],"languages":[{"name":"Chinese","url":"/datasets/language/chinese"}],"variants":["ASR-RAMC-BIGCCSC: A CHINESE CONVERSATIONAL SPEECH CORPUS"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}