{"url":"/dataset/media","name":"MEDIA","full_name":"MEDIA","description_markdown":"The **MEDIA** French corpus is dedicated to semantic extraction from speech in a context of human/machine dialogues. The corpus has manual transcription and conceptual annotation of dialogues from 250 speakers. It is split into the following three parts : (1) the training set (720 dialogues, 12K sentences), (2) the development set (79 dialogues, 1.3K sentences, and (3) the test set (200 dialogues, 3K sentences).\n\nSource: [Dialogue history integration into end-to-end signal-to-concept spoken language understanding systems](https://arxiv.org/abs/2002.06012)\nImage Source: [http://www.lrec-conf.org/proceedings/lrec2004/pdf/356.pdf](http://www.lrec-conf.org/proceedings/lrec2004/pdf/356.pdf)","description_withheld":null,"homepage":"http://www.lrec-conf.org/proceedings/lrec2004/pdf/356.pdf","introduced_date":"2004-01-01","introduced_date_note":null,"introduced_by":{"paper":null,"title":"The French MEDIA/EVALDA Project: the Evaluation of the Understanding Capability of Spoken Language Dialogue Systems","first_author":null,"url":"http://www.lrec-conf.org/proceedings/lrec2004/summaries/356.htm"},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"},{"name":"Audio","url":"/datasets/modality/audio"}],"tasks":[{"name":"Named Entity Recognition (NER)","url":"/task/named-entity-recognition-ner","datasets_with_task":"/datasets/task/named-entity-recognition-ner"},{"name":"Slot Filling","url":"/task/slot-filling","datasets_with_task":"/datasets/task/slot-filling"},{"name":"Spoken Language Understanding","url":"/task/spoken-language-understanding","datasets_with_task":"/datasets/task/spoken-language-understanding"}],"languages":[{"name":"English","url":"/datasets/language/english"},{"name":"French","url":"/datasets/language/french"}],"variants":["MEDIA"],"data_loaders":[],"num_papers_in_archive":6,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}