{"url":"/task/dialect-identification","name":"Dialect Identification","slug":"dialect-identification","description_markdown":"Dialectal Arabic Identification","categories":[{"name":"Audio","url":"/area/audio"},{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":189,"papers_with_code":33,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":3,"subtasks":0,"parent_tasks":1},"benchmarks":[],"datasets":[{"url":"/dataset/arsarcasm-v2","name":"ArSarcasm-v2","full_name":"","num_papers_in_archive":15},{"url":"/dataset/arsarcasm","name":"ArSarcasm","full_name":"","num_papers_in_archive":14},{"url":"/dataset/frecdo","name":"FreCDo","full_name":"French cross-domain","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[{"url":"/task/language-identification","name":"Language Identification"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":33,"tagged_in_all":189,"items":[{"url":"/paper/glotlid-language-identification-for-low","title":"GlotLID: Language Identification for Low-Resource Languages","date":"2023-10-24","arxiv_id":"2310.16248","repositories_listed":3,"syntology":null},{"url":"/paper/resource-aware-arabic-llm-creation-model","title":"Resource-Aware Arabic LLM Creation: Model Adaptation, Integration, and Multi-Domain Testing","date":"2024-12-23","arxiv_id":"2412.17548","repositories_listed":1,"syntology":null},{"url":"/paper/multi-dialect-vietnamese-task-dataset","title":"Multi-Dialect Vietnamese: Task, Dataset, Baseline Models and Challenges","date":"2024-10-04","arxiv_id":"2410.03458","repositories_listed":1,"syntology":null},{"url":"/paper/sebastian-basti-wastl-recognizing-named","title":"Sebastian, Basti, Wastl?! Recognizing Named Entities in Bavarian Dialectal Data","date":"2024-03-19","arxiv_id":"2403.12749","repositories_listed":1,"syntology":null},{"url":"/paper/artst-arabic-text-and-speech-transformer","title":"ArTST: Arabic Text and Speech Transformer","date":"2023-10-25","arxiv_id":"2310.16621","repositories_listed":1,"syntology":null},{"url":"/paper/arabic-dialect-identification-under-scrutiny","title":"Arabic Dialect Identification under Scrutiny: Limitations of Single-label Classification","date":"2023-10-20","arxiv_id":"2310.13661","repositories_listed":1,"syntology":null},{"url":"/paper/aldi-quantifying-the-arabic-level-of","title":"ALDi: Quantifying the Arabic Level of Dialectness of Text","date":"2023-10-20","arxiv_id":"2310.13747","repositories_listed":1,"syntology":null},{"url":"/paper/dada-dialect-adaptation-via-dynamic","title":"DADA: Dialect Adaptation via Dynamic Aggregation of Linguistic Rules","date":"2023-05-22","arxiv_id":"2305.13406","repositories_listed":1,"syntology":null},{"url":"/paper/north-sami-dialect-identification-with-self","title":"North Sámi Dialect Identification with Self-supervised Speech Models","date":"2023-05-19","arxiv_id":"2305.11864","repositories_listed":1,"syntology":null},{"url":"/paper/a-parameter-efficient-learning-approach-to","title":"A Parameter-Efficient Learning Approach to Arabic Dialect Identification with Pre-Trained General-Purpose Speech Model","date":"2023-05-18","arxiv_id":"2305.11244","repositories_listed":1,"syntology":null},{"url":"/paper/two-stage-pipeline-for-multilingual-dialect","title":"Two-stage Pipeline for Multilingual Dialect Detection","date":"2023-03-06","arxiv_id":"2303.03487","repositories_listed":1,"syntology":null},{"url":"/paper/frecdo-a-large-corpus-for-french-cross-domain","title":"FreCDo: A Large Corpus for French Cross-Domain Dialect Identification","date":"2022-12-15","arxiv_id":"2212.07707","repositories_listed":1,"syntology":null},{"url":"/paper/a-benchmark-study-of-contrastive-learning-for","title":"A Benchmark Study of Contrastive Learning for Arabic Social Meaning","date":"2022-10-22","arxiv_id":"2210.12314","repositories_listed":1,"syntology":null},{"url":"/paper/nadi-2022-the-third-nuanced-arabic-dialect","title":"NADI 2022: The Third Nuanced Arabic Dialect Identification Shared Task","date":"2022-10-18","arxiv_id":"2210.09582","repositories_listed":1,"syntology":null},{"url":"/paper/findings-of-the-vardial-evaluation-campaign-2","title":"Findings of the VarDial Evaluation Campaign 2022","date":"2022-10-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/italian-language-and-dialect-identification","title":"Italian Language and Dialect Identification and Regional French Variety Detection using Adaptive Naive Bayes","date":"2022-10-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/distilling-the-knowledge-of-romanian-berts","title":"Distilling the Knowledge of Romanian BERTs Using Multiple Teachers","date":"2021-12-23","arxiv_id":"2112.12650","repositories_listed":1,"syntology":null},{"url":"/paper/finnish-dialect-identification-the-effect-of-1","title":"Finnish Dialect Identification: The Effect of Audio and Text","date":"2021-11-06","arxiv_id":"2111.03800","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-multi-scale-convolution-for-dialect","title":"Dynamic Multi-scale Convolution for Dialect Identification","date":"2021-08-02","arxiv_id":"2108.07787","repositories_listed":1,"syntology":null},{"url":"/paper/aracovid19-mfh-arabic-covid-19-multi-label","title":"AraCOVID19-MFH: Arabic COVID-19 Multi-label Fake News and Hate Speech Detection Dataset","date":"2021-05-07","arxiv_id":"2105.03143","repositories_listed":1,"syntology":null},{"url":"/paper/comparing-the-performance-of-cnns-and-shallow","title":"Comparing the Performance of CNNs and Shallow Models for Language Identification","date":"2021-04-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/nadi-2021-the-second-nuanced-arabic-dialect","title":"NADI 2021: The Second Nuanced Arabic Dialect Identification Shared Task","date":"2021-03-04","arxiv_id":"2103.08466","repositories_listed":1,"syntology":null},{"url":"/paper/adapting-marbert-for-improved-arabic-dialect","title":"Adapting MARBERT for Improved Arabic Dialect Identification: Submission to the NADI 2021 Shared Task","date":"2021-03-01","arxiv_id":"2103.01065","repositories_listed":1,"syntology":null},{"url":"/paper/arabic-dialect-identification-using-bert-fine","title":"Arabic Dialect Identification Using BERT Fine-Tuning","date":"2020-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/a-dual-encoding-system-for-dialect","title":"A dual-encoding system for dialect classification","date":"2020-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/toward-micro-dialect-identification-in","title":"Toward Micro-Dialect Identification in Diaglossic and Code-Switched Environments","date":"2020-10-10","arxiv_id":"2010.04900","repositories_listed":1,"syntology":null},{"url":"/paper/the-unreasonable-effectiveness-of-machine","title":"The Unreasonable Effectiveness of Machine Learning in Moldavian versus Romanian Dialect Identification","date":"2020-07-30","arxiv_id":"2007.15700","repositories_listed":1,"syntology":null},{"url":"/paper/multi-dialect-arabic-bert-for-country-level","title":"Multi-Dialect Arabic BERT for Country-Level Dialect Identification","date":"2020-07-10","arxiv_id":"2007.05612","repositories_listed":1,"syntology":null},{"url":"/paper/camel-tools-an-open-source-python-toolkit-for","title":"CAMeL Tools: An Open Source Python Toolkit for Arabic Natural Language Processing","date":"2020-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/speech-recognition-challenge-in-the-wild","title":"Speech Recognition Challenge in the Wild: Arabic MGB-3","date":"2017-09-21","arxiv_id":"1709.07276","repositories_listed":1,"syntology":null}],"syntology_records":0,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}