{"url":"/dataset/magicdata-ramc","name":"MagicData-RAMC","full_name":null,"description_markdown":"The MagicData-RAMC corpus contains 180 hours of conversational speech data recorded from native speakers of Mandarin Chinese over mobile phones with a sampling rate of 16 kHz. The dialogs in the dialogs are classified into 15 diversified domains and tagged with topic labels, ranging from science and technology to ordinary life. Accurate transcription and precise speaker voice activity timestamps are manually labeled for each sample. Speakers' detailed information is also provided.","description_withheld":null,"homepage":"https://www.magicdatatech.com/datasets/mdt2021s003-1647827542","introduced_date":"2022-03-31","introduced_date_note":null,"introduced_by":{"paper":"/paper/open-source-magicdata-ramc-a-rich-annotated","title":"Open Source MagicData-RAMC: A Rich Annotated Mandarin Conversational(RAMC) Speech Dataset","first_author":"Zehui Yang","url":null},"license":{"name":"CC BY-NC 4.0","url":"https://creativecommons.org/licenses/by-nc/4.0/"},"modalities":[{"name":"Speech","url":"/datasets/modality/speech"}],"tasks":[{"name":"Speech Recognition","url":"/task/speech-recognition","datasets_with_task":"/datasets/task/speech-recognition"},{"name":"Speaker Diarization","url":"/task/speaker-diarization","datasets_with_task":"/datasets/task/speaker-diarization"},{"name":"Automatic Speech Recognition (ASR)","url":"/task/automatic-speech-recognition","datasets_with_task":"/datasets/task/automatic-speech-recognition"}],"languages":[{"name":"Chinese","url":"/datasets/language/chinese"},{"name":"Mandarin Chinese","url":"/datasets/language/mandarin-chinese"}],"variants":["MagicData-RAMC"],"data_loaders":[],"num_papers_in_archive":12,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}