{"url":"/task/language-identification","name":"Language Identification","slug":"language-identification","description_markdown":"Language identification is the task of determining the language of a text.","categories":[{"name":"Audio","url":"/area/audio"},{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":794,"papers_with_code":143,"benchmarks":6,"benchmark_tables_in_archive":6,"benchmark_tables_shown":6,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":20,"subtasks":2,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/language-identification-on-voxlingua107-1","slug":"language-identification-on-voxlingua107-1","dataset":"VOXLINGUA107","dataset_url":"/dataset/voxlingua107","rows_in_archive":2,"metrics":["Error rate"],"first_row_in_archive_order":{"model":"XLS-R","paper_title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","paper_url":"/paper/xls-r-self-supervised-cross-lingual-speech","paper_date":"2021-11-17","arxiv_id":"2111.09296","code_links":[{"title":"pytorch/fairseq","url":"https://github.com/pytorch/fairseq"},{"title":"gatech-eic/s3-router","url":"https://github.com/gatech-eic/s3-router"}],"syntology":null}},{"leaderboard":"/sota/language-identification-on-glotlid-c","slug":"language-identification-on-glotlid-c","dataset":"GlotLID-C","dataset_url":null,"rows_in_archive":1,"metrics":["Macro F1"],"first_row_in_archive_order":{"model":"GlotLID","paper_title":"GlotLID: Language Identification for Low-Resource Languages","paper_url":"/paper/glotlid-language-identification-for-low","paper_date":"2023-10-24","arxiv_id":"2310.16248","code_links":[{"title":"cisnlp/glotlid","url":"https://github.com/cisnlp/glotlid"},{"title":"cisnlp/glotstorybook","url":"https://github.com/cisnlp/glotstorybook"},{"title":"cisnlp/glotsparse","url":"https://github.com/cisnlp/glotsparse"}],"syntology":null}},{"leaderboard":"/sota/language-identification-on-nordic-langid","slug":"language-identification-on-nordic-langid","dataset":"Nordic Language Identification","dataset_url":"/dataset/nordic-langid","rows_in_archive":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"FastText","paper_title":"Discriminating Between Similar Nordic Languages","paper_url":"/paper/discriminating-between-similar-nordic","paper_date":"2020-12-11","arxiv_id":"2012.06431","code_links":[{"title":"renhaa/NordicDSL","url":"https://github.com/renhaa/NordicDSL"},{"title":"StrombergNLP/NordicDSL","url":"https://github.com/StrombergNLP/NordicDSL"}],"syntology":null}},{"leaderboard":"/sota/language-identification-on-opensubtitles","slug":"language-identification-on-opensubtitles","dataset":"OpenSubtitles","dataset_url":"/dataset/opensubtitles","rows_in_archive":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Apple bi-LSTM","paper_title":"A reproduction of Apple's bi-directional LSTM models for language identification in short strings","paper_url":"/paper/a-reproduction-of-apple-s-bi-directional-lstm","paper_date":"2021-02-11","arxiv_id":"2102.06282","code_links":[{"title":"AU-DIS/LSTM_langid","url":"https://github.com/AU-DIS/LSTM_langid"}],"syntology":null}},{"leaderboard":"/sota/language-identification-on-universal","slug":"language-identification-on-universal","dataset":"Universal Dependencies","dataset_url":"/dataset/universal-dependencies","rows_in_archive":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Apple bi-LSTM","paper_title":"A reproduction of Apple's bi-directional LSTM models for language identification in short strings","paper_url":"/paper/a-reproduction-of-apple-s-bi-directional-lstm","paper_date":"2021-02-11","arxiv_id":"2102.06282","code_links":[{"title":"AU-DIS/LSTM_langid","url":"https://github.com/AU-DIS/LSTM_langid"}],"syntology":null}},{"leaderboard":"/sota/language-identification-on-voxforge","slug":"language-identification-on-voxforge","dataset":"VoxForge","dataset_url":"/dataset/voxforge","rows_in_archive":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"ConformerG-P","paper_title":"BigSSL: Exploring the Frontier of Large-Scale Semi-Supervised Learning for Automatic Speech Recognition","paper_url":"/paper/bigssl-exploring-the-frontier-of-large-scale","paper_date":"2021-09-27","arxiv_id":"2109.13226","code_links":[],"syntology":null}}],"datasets":[{"url":"/dataset/universal-dependencies","name":"Universal Dependencies","full_name":"","num_papers_in_archive":520},{"url":"/dataset/common-voice","name":"Common Voice","full_name":"Common Voice","num_papers_in_archive":449},{"url":"/dataset/opensubtitles","name":"OpenSubtitles","full_name":"","num_papers_in_archive":214},{"url":"/dataset/olid","name":"OLID","full_name":"Offensive Language Identification Dataset","num_papers_in_archive":152},{"url":"/dataset/conan","name":"CONAN","full_name":"COunter NArratives through Nichesourcing","num_papers_in_archive":27},{"url":"/dataset/moroco","name":"MOROCO","full_name":"MOldavian and ROmanian Dialectal COrpus","num_papers_in_archive":22},{"url":"/dataset/dakshina","name":"Dakshina","full_name":"Dakshina","num_papers_in_archive":14},{"url":"/dataset/voxforge","name":"VoxForge","full_name":"VoxForge","num_papers_in_archive":11},{"url":"/dataset/hindencorp","name":"HindEnCorp","full_name":"","num_papers_in_archive":5},{"url":"/dataset/ogtd","name":"OGTD","full_name":"Offensive Greek Tweet Dataset","num_papers_in_archive":3},{"url":"/dataset/wili-2018","name":"WiLI-2018","full_name":"","num_papers_in_archive":3},{"url":"/dataset/udhr-lid","name":"udhr-lid","full_name":"","num_papers_in_archive":2},{"url":"/dataset/voxlingua107","name":"VOXLINGUA107","full_name":"","num_papers_in_archive":2},{"url":"/dataset/glotsparse","name":"GlotSparse","full_name":"","num_papers_in_archive":1},{"url":"/dataset/glotstorybook","name":"GlotStoryBook","full_name":"","num_papers_in_archive":1},{"url":"/dataset/l3cube-mahacorpus","name":"L3Cube-MahaCorpus","full_name":"","num_papers_in_archive":1},{"url":"/dataset/nli-pt","name":"NLI-PT","full_name":"","num_papers_in_archive":1},{"url":"/dataset/nordic-langid","name":"Nordic Language Identification","full_name":"","num_papers_in_archive":1},{"url":"/dataset/tugebic","name":"TuGebic","full_name":"A Turkish-German Bilingual Code-Switching Corpus","num_papers_in_archive":1},{"url":"/dataset/english-pashto-language-dataset-epld","name":"English-Pashto Language Dataset (EPLD)","full_name":"","num_papers_in_archive":0}],"subtasks":[{"url":"/task/dialect-identification","name":"Dialect Identification"},{"url":"/task/native-language-identification","name":"Native Language Identification"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":143,"tagged_in_all":794,"items":[{"url":"/paper/scaling-speech-technology-to-1000-languages-1","title":"Scaling Speech Technology to 1,000+ Languages","date":"2023-05-22","arxiv_id":"2305.13516","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/speechbrain-a-general-purpose-speech-toolkit","title":"SpeechBrain: A General-Purpose Speech Toolkit","date":"2021-06-08","arxiv_id":"2106.04624","repositories_listed":4,"syntology":{"n":7,"n_ran":0,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/the-wili-benchmark-dataset-for-written","title":"The WiLI benchmark dataset for written language identification","date":"2018-01-23","arxiv_id":"1801.07779","repositories_listed":4,"syntology":null},{"url":"/paper/glotlid-language-identification-for-low","title":"GlotLID: Language Identification for Low-Resource Languages","date":"2023-10-24","arxiv_id":"2310.16248","repositories_listed":3,"syntology":null},{"url":"/paper/cosyvoice-3-towards-in-the-wild-speech","title":"CosyVoice 3: Towards In-the-wild Speech Generation via Scaling-up and Post-training","date":"2025-05-23","arxiv_id":"2505.17589","repositories_listed":2,"syntology":{"n":4,"n_ran":1,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/glotcc-an-open-broad-coverage-commoncrawl","title":"GlotCC: An Open Broad-Coverage CommonCrawl Corpus and Pipeline for Minority Languages","date":"2024-10-31","arxiv_id":"2410.23825","repositories_listed":2,"syntology":null},{"url":"/paper/from-n-grams-to-pre-trained-multilingual","title":"From N-grams to Pre-trained Multilingual Models For Language Identification","date":"2024-10-11","arxiv_id":"2410.08728","repositories_listed":2,"syntology":null},{"url":"/paper/kinit-at-semeval-2024-task-8-fine-tuned-llms","title":"KInIT at SemEval-2024 Task 8: Fine-tuned LLMs for Multilingual Machine-Generated Text Detection","date":"2024-02-21","arxiv_id":"2402.13671","repositories_listed":2,"syntology":null},{"url":"/paper/xls-r-self-supervised-cross-lingual-speech","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","date":"2021-11-17","arxiv_id":"2111.09296","repositories_listed":2,"syntology":null},{"url":"/paper/discriminating-between-similar-nordic","title":"Discriminating Between Similar Nordic Languages","date":"2020-12-11","arxiv_id":"2012.06431","repositories_listed":2,"syntology":null},{"url":"/paper/adelaidecyc-at-semeval-2020-task-12-ensemble","title":"AdelaideCyC at SemEval-2020 Task 12: Ensemble of Classifiers for Offensive Language Detection in Social Media","date":"2020-12-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/cybertronics-at-semeval-2020-task-12","title":"CyberTronics at SemEval-2020 Task 12: Multilingual Offensive Language Identification over Social Media","date":"2020-12-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/voxlingua107-a-dataset-for-spoken-language-1","title":"VoxLingua107: a Dataset for Spoken Language Recognition","date":"2020-11-25","arxiv_id":"2011.12998","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/common-voice-a-massively-multilingual-speech","title":"Common Voice: A Massively-Multilingual Speech Corpus","date":"2019-12-13","arxiv_id":"1912.06670","repositories_listed":2,"syntology":null},{"url":"/paper/speech-vgg-a-deep-feature-extractor-for","title":"Word-level Embeddings for Cross-Task Transfer Learning in Speech Processing","date":"2019-10-22","arxiv_id":"1910.09909","repositories_listed":2,"syntology":null},{"url":"/paper/semeval-2019-task-6-identifying-and-1","title":"SemEval-2019 Task 6: Identifying and Categorizing Offensive Language in Social Media (OffensEval)","date":"2019-03-19","arxiv_id":"1903.08983","repositories_listed":2,"syntology":{"n":10,"n_ran":0,"n_unverified":10,"n_pointer_only":0}},{"url":"/paper/universal-dependency-parsing-for-hindi","title":"Universal Dependency Parsing for Hindi-English Code-switching","date":"2018-04-16","arxiv_id":"1804.05868","repositories_listed":2,"syntology":null},{"url":"/paper/advancing-uto-aztecan-language-technologies-a","title":"Advancing Uto-Aztecan Language Technologies: A Case Study on the Endangered Comanche Language","date":"2025-05-10","arxiv_id":"2505.18159","repositories_listed":1,"syntology":null},{"url":"/paper/kreyolid-from-language-identification-towards","title":"KréyoLID From Language Identification Towards Language Mining","date":"2025-03-09","arxiv_id":"2503.06547","repositories_listed":1,"syntology":null},{"url":"/paper/english-please-evaluating-machine-translation","title":"English Please: Evaluating Machine Translation with Large Language Models for Multilingual Bug Reports","date":"2025-02-20","arxiv_id":"2502.14338","repositories_listed":1,"syntology":null},{"url":"/paper/multi-label-scandinavian-language","title":"Multi-label Scandinavian Language Identification (SLIDE)","date":"2025-02-10","arxiv_id":"2502.06692","repositories_listed":1,"syntology":null},{"url":"/paper/is-it-navajo-accurate-language-detection-in","title":"Is It Navajo? Accurate Language Detection in Endangered Athabaskan Languages","date":"2025-01-27","arxiv_id":"2501.15773","repositories_listed":1,"syntology":null},{"url":"/paper/fleurs-slu-a-massively-multilingual-benchmark","title":"Fleurs-SLU: A Massively Multilingual Benchmark for Spoken Language Understanding","date":"2025-01-10","arxiv_id":"2501.06117","repositories_listed":1,"syntology":null},{"url":"/paper/afrihubert-a-self-supervised-speech","title":"AfriHuBERT: A self-supervised speech representation model for African languages","date":"2024-09-30","arxiv_id":"2409.20201","repositories_listed":1,"syntology":null},{"url":"/paper/improving-multilingual-asr-in-the-wild-using","title":"Improving Multilingual ASR in the Wild Using Simple N-best Re-ranking","date":"2024-09-27","arxiv_id":"2409.18428","repositories_listed":1,"syntology":null},{"url":"/paper/language-informed-beam-search-decoding-for","title":"Language-Informed Beam Search Decoding for Multilingual Machine Translation","date":"2024-08-11","arxiv_id":"2408.05738","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/speech-massive-a-multilingual-speech-dataset","title":"Speech-MASSIVE: A Multilingual Speech Dataset for SLU and Beyond","date":"2024-08-07","arxiv_id":"2408.03900","repositories_listed":1,"syntology":null},{"url":"/paper/script-agnostic-language-identification","title":"Script-Agnostic Language Identification","date":"2024-06-25","arxiv_id":"2406.17901","repositories_listed":1,"syntology":null},{"url":"/paper/an-initial-investigation-of-language","title":"An Initial Investigation of Language Adaptation for TTS Systems under Low-resource Scenarios","date":"2024-06-13","arxiv_id":"2406.08911","repositories_listed":1,"syntology":null},{"url":"/paper/masklid-code-switching-language","title":"MaskLID: Code-Switching Language Identification through Iterative Masking","date":"2024-06-10","arxiv_id":"2406.06263","repositories_listed":1,"syntology":null}],"syntology_records":6,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}