{"url":"/dataset/evi","name":"EVI","full_name":null,"description_markdown":"The EVI dataset is a challenging, multilingual spoken-dialogue dataset with 5,506 dialogues in English, Polish, and French. The dataset can be used to develop and benchmark conversational systems for user authentication tasks, i.e. speaker enrolment (E), speaker verification (V), speaker identification (I).\r\n\r\nThe dataset contains the audio data, machine transcriptions, and target identity for each dialogue,  and the knowledge-base with personal information (postcode, name, and date of birth) for each identity.  The dataset can be used with both text-independent biometric or knowledge-based authentication (KBA) tasks.","description_withheld":null,"homepage":"https://github.com/PolyAI-LDN/evi-paper","introduced_date":"2022-04-28","introduced_date_note":null,"introduced_by":null,"license":{"name":"CC-BY-4.0 license","url":"https://github.com/PolyAI-LDN/evi-paper/blob/main/LICENCE"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"},{"name":"Tabular","url":"/datasets/modality/tabular"},{"name":"Speech","url":"/datasets/modality/speech"},{"name":"Dialog","url":"/datasets/modality/dialog"}],"tasks":[{"name":"Speaker Verification","url":"/task/speaker-verification","datasets_with_task":"/datasets/task/speaker-verification"},{"name":"Speaker Identification","url":"/task/speaker-identification","datasets_with_task":"/datasets/task/speaker-identification"}],"languages":[{"name":"English","url":"/datasets/language/english"},{"name":"French","url":"/datasets/language/french"},{"name":"Polish","url":"/datasets/language/polish"}],"variants":["EVI","EVI en-GB","EVI pl-PL","EVI fr-FR"],"data_loaders":[{"repo":"https://github.com/PolyAI-LDN/evi-paper","url":"https://github.com/PolyAI-LDN/evi-paper","frameworks":[]}],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/speaker-identification-on-evi-en-gb-1","task":"Speaker Identification","dataset_variant":"EVI en-GB","rows":1,"metrics":["Top-1 (%)"],"first_row_in_archive_order":{"model":"Fuzzy Retrieval","paper":"/paper/evi-multilingual-spoken-dialogue-tasks-and-1","metrics":{"Top-1 (%)":"67.77"},"code_links":[{"title":"PolyAI-LDN/evi-paper","url":"https://github.com/PolyAI-LDN/evi-paper"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speaker-identification-on-evi-fr-fr","task":"Speaker Identification","dataset_variant":"EVI fr-FR","rows":1,"metrics":["Top-1 (%)"],"first_row_in_archive_order":{"model":"Fuzzy Retrieval","paper":"/paper/evi-multilingual-spoken-dialogue-tasks-and-1","metrics":{"Top-1 (%)":"80.83"},"code_links":[{"title":"PolyAI-LDN/evi-paper","url":"https://github.com/PolyAI-LDN/evi-paper"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speaker-identification-on-evi-pl-pl","task":"Speaker Identification","dataset_variant":"EVI pl-PL","rows":1,"metrics":["Top-1 (%)"],"first_row_in_archive_order":{"model":"Fuzzy Retrieval","paper":"/paper/evi-multilingual-spoken-dialogue-tasks-and-1","metrics":{"Top-1 (%)":"95.13"},"code_links":[{"title":"PolyAI-LDN/evi-paper","url":"https://github.com/PolyAI-LDN/evi-paper"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/evi-multilingual-spoken-dialogue-tasks-and-1","title":"EVI: Multilingual Spoken Dialogue Tasks and Dataset for Knowledge-Based Enrolment, Verification, and Identification","date":"2022-04-28","rows_on_this_dataset":3,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}