{"url":"/dataset/arzen","name":"ArzEn","full_name":"Corpus of Egyptian Arabic-English Code-switching","description_markdown":"**Corpus of Egyptian Arabic-English Code-switching (ArzEn)** is a spontaneous conversational speech corpus, obtained through informal interviews held at the German University in Cairo. The participants discussed broad topics, including education, hobbies, work, and life experiences. The corpus currently contains 12 hours of speech, having 6,216 utterances. The recordings were transcribed and translated into monolingual Egyptian Arabic and monolingual English.\r\n\r\nSource: [https://arxiv.org/pdf/2211.16319v1.pdf](https://arxiv.org/pdf/2211.16319v1.pdf)","description_withheld":null,"homepage":"https://sites.google.com/view/arzen-corpus/home","introduced_date":"2022-11-22","introduced_date_note":null,"introduced_by":{"paper":"/paper/benchmarking-evaluation-metrics-for-code","title":"Benchmarking Evaluation Metrics for Code-Switching Automatic Speech Recognition","first_author":"Injy Hamed","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Speech Recognition","url":"/task/speech-recognition","datasets_with_task":"/datasets/task/speech-recognition"},{"name":"Automatic Speech Recognition (ASR)","url":"/task/automatic-speech-recognition","datasets_with_task":"/datasets/task/automatic-speech-recognition"},{"name":"ArzEn Code-switched Translation to eng","url":"/task/arzen-code-switched-translation-to-eng","datasets_with_task":"/datasets/task/arzen-code-switched-translation-to-eng"},{"name":"ArzEn Code-switched Translation to ara","url":"/task/arzen-code-switched-translation-to-ara","datasets_with_task":"/datasets/task/arzen-code-switched-translation-to-ara"},{"name":"ArzEn Speech Recognition","url":"/task/arzen-speech-recognition","datasets_with_task":"/datasets/task/arzen-speech-recognition"},{"name":"Transliteration","url":"/task/transliteration","datasets_with_task":"/datasets/task/transliteration"}],"languages":[{"name":"English","url":"/datasets/language/english"},{"name":"Egyptian Arabic","url":"/datasets/language/egyptian-arabic"}],"variants":["ArzEn"],"data_loaders":[],"num_papers_in_archive":3,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}