{"url":"/dataset/egyptian-arabic-segmentation-dataset","name":"Egyptian Arabic Segmentation Dataset","full_name":null,"description_markdown":"Contains 350 tweets with more than 8,000 words including 3,000 unique words written in Egyptian dialect. The tweets have much dialectal content covering most of dialectal Egyptian phonological, morphological, and syntactic phenomena. It also includes Twitter-specific aspects of the text, such as #hashtags, @mentions, emoticons and URLs.\r\n\r\nSource: [A Neural Architecture for Dialectal Arabic Segmentation](/paper/a-neural-architecture-for-dialectal-arabic)","description_withheld":null,"homepage":"https://alt.qcri.org/resources/da_resources/","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/a-neural-architecture-for-dialectal-arabic","title":"A Neural Architecture for Dialectal Arabic Segmentation","first_author":"Younes Samih","url":null},"license":null,"modalities":[],"tasks":[{"name":"Part-Of-Speech Tagging","url":"/task/part-of-speech-tagging","datasets_with_task":"/datasets/task/part-of-speech-tagging"},{"name":"Morphological Analysis","url":"/task/morphological-analysis","datasets_with_task":"/datasets/task/morphological-analysis"}],"languages":[{"name":"Arabic","url":"/datasets/language/arabic"}],"variants":["Egyptian Arabic Segmentation Dataset"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}