{"url":"/dataset/catt-dataset","name":"CATT","full_name":"CATT Arabic Diacritization Benchmark Dataset","description_markdown":"The CATT benchmark dataset comprises 742 sentences, which were scraped from an internet news source in 2023.\r\nIt covers multiple topics including science and technology, economics, politics, sports, arts, and culture.\r\nIt was manually diacritized by two expert native Arabic speakers and then validated by a third expert.\r\nThis dataset contains names of people and places in both Arabic and English.\r\nAs for the English names, they are written in Arabic letters and diacritized based on their pronunciation.\r\nAlso, the numbers in the sentences are written in textual form rather than the numeric form which helps in evaluating the models without the need for a text normalizer (TN).","description_withheld":null,"homepage":"https://github.com/abjadai/catt","introduced_date":"2024-07-03","introduced_date_note":null,"introduced_by":{"paper":"/paper/catt-character-based-arabic-tashkeel","title":"CATT: Character-based Arabic Tashkeel Transformer","first_author":"Faris Alasmary","url":null},"license":{"name":"CC BY-NC-SA","url":"https://creativecommons.org/licenses/by-nc/4.0/"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Arabic Text Diacritization","url":"/task/arabic-text-diacritization","datasets_with_task":"/datasets/task/arabic-text-diacritization"}],"languages":[{"name":"Arabic","url":"/datasets/language/arabic"}],"variants":["CATT"],"data_loaders":[],"num_papers_in_archive":3,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/arabic-text-diacritization-on-catt-dataset","task":"Arabic Text Diacritization","dataset_variant":"CATT","rows":12,"metrics":["DER(%)","WER (%)"],"first_row_in_archive_order":{"model":"CATT ED","paper":"/paper/catt-character-based-arabic-tashkeel","metrics":{"DER(%)":"8.624","WER (%)":"34.191"},"code_links":[{"title":"abjadai/catt","url":"https://github.com/abjadai/catt"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/catt-character-based-arabic-tashkeel","title":"CATT: Character-based Arabic Tashkeel Transformer","date":"2024-07-03","rows_on_this_dataset":9,"code_links":1,"syntology":null},{"paper":"/paper/deep-diacritization-efficient-hierarchical","title":"Deep Diacritization: Efficient Hierarchical Recurrence for Improved Arabic Diacritization","date":"2020-11-01","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/farasa-a-fast-and-furious-segmenter-for","title":"Farasa: A Fast and Furious Segmenter for Arabic","date":"2016-06-01","rows_on_this_dataset":1,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}