{"url":"/dataset/fmfcc-a","name":"FMFCC-A","full_name":null,"description_markdown":"FMFCC-A is a large publicly-available Mandarin dataset for synthetic speech detection, which contains 40,000 synthesized Mandarin utterances that generated by 11 Mandarin TTS systems and two Mandarin VC systems, and 10,000 genuine Mandarin utterance collected from 58 speakers. The FMFCCA dataset is divided into the training, development and evaluation sets, which are used for the research of detection of synthesised Mandarin speech under various previously unknown speech synthesis systems or audio post-processing operations.","description_withheld":null,"homepage":"https://github.com/Amforever/FMFCC-A","introduced_date":"2021-10-18","introduced_date_note":null,"introduced_by":{"paper":"/paper/fmfcc-a-a-challenging-mandarin-dataset-for","title":"FMFCC-A: A Challenging Mandarin Dataset for Synthetic Speech Detection","first_author":"Zhenyu Zhang","url":null},"license":null,"modalities":[{"name":"Speech","url":"/datasets/modality/speech"}],"tasks":[],"languages":[{"name":"Mandarin Chinese","url":"/datasets/language/mandarin-chinese"}],"variants":["FMFCC-A"],"data_loaders":[],"num_papers_in_archive":6,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}