{"url":"/dataset/mfaq","name":"MFAQ","full_name":null,"description_markdown":"** MFAQ** is a multilingual FAQ dataset publicly available. It contains around 6M FAQ pairs from the web, in 21 different languages. Although this is significantly larger than existing FAQ retrieval datasets, it comes with its own challenges: duplication of content and uneven distribution of topics.","description_withheld":null,"homepage":"https://github.com/clips/mfaq","introduced_date":"2021-09-27","introduced_date_note":null,"introduced_by":{"paper":"/paper/mfaq-a-multilingual-faq-dataset","title":"MFAQ: a Multilingual FAQ Dataset","first_author":"Maxime De Bruyn","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[],"languages":[],"variants":["MFAQ"],"data_loaders":[],"num_papers_in_archive":3,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}