{"url":"/task/spoken-language-identification","name":"Spoken language identification","slug":"spoken-language-identification","description_markdown":"Identify the language being spoken from an audio input only.","categories":[{"name":"Speech","url":"/area/speech"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":51,"papers_with_code":13,"benchmarks":12,"benchmark_tables_in_archive":12,"benchmark_tables_shown":12,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":4,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/spoken-language-identification-on-lre07","slug":"spoken-language-identification-on-lre07","dataset":"LRE07","dataset_url":null,"rows_in_archive":9,"metrics":["3 sec","10 sec","30 sec","Average"],"first_row_in_archive_order":{"model":"CNN-LDE","paper_title":"VOXLINGUA107: A DATASET FOR SPOKEN LANGUAGE RECOGNITION","paper_url":"/paper/voxlingua107-a-dataset-for-spoken-language","paper_date":"2020-11-25","arxiv_id":"2011.12998","code_links":[],"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}}},{"leaderboard":"/sota/spoken-language-identification-on-voxforge-1","slug":"spoken-language-identification-on-voxforge-1","dataset":"VoxForge European","dataset_url":"/dataset/voxforge","rows_in_archive":5,"metrics":["Accuracy (%)"],"first_row_in_archive_order":{"model":"2D ConvNet(MixUp=YES)","paper_title":"Spoken Language Identification using ConvNets","paper_url":"/paper/spoken-language-identification-using-convnets","paper_date":"2019-10-09","arxiv_id":"1910.04269","code_links":[],"syntology":null}},{"leaderboard":"/sota/spoken-language-identification-on-youtube","slug":"spoken-language-identification-on-youtube","dataset":"YouTube News dataset (No Noise)","dataset_url":null,"rows_in_archive":5,"metrics":["Accuracy ","F1 Score"],"first_row_in_archive_order":{"model":"CRNN","paper_title":"Is Attention always needed? A Case Study on Language Identification from Speech","paper_url":"/paper/is-attention-always-needed-a-case-study-on","paper_date":"2021-10-05","arxiv_id":"2110.03427","code_links":[],"syntology":null}},{"leaderboard":"/sota/spoken-language-identification-on-youtube-1","slug":"spoken-language-identification-on-youtube-1","dataset":"YouTube News dataset (White Noise)","dataset_url":null,"rows_in_archive":5,"metrics":["Accuracy ","F1 Score"],"first_row_in_archive_order":{"model":"CRNN","paper_title":"Is Attention always needed? A Case Study on Language Identification from Speech","paper_url":"/paper/is-attention-always-needed-a-case-study-on","paper_date":"2021-10-05","arxiv_id":"2110.03427","code_links":[],"syntology":null}},{"leaderboard":"/sota/spoken-language-identification-on-1","slug":"spoken-language-identification-on-1","dataset":"Untranscribed mixed-speech dataset","dataset_url":null,"rows_in_archive":4,"metrics":["ACC","PRC","RCL"],"first_row_in_archive_order":{"model":"SVM","paper_title":"Automatic Dialect Detection in Arabic Broadcast Speech","paper_url":"/paper/automatic-dialect-detection-in-arabic","paper_date":"2015-09-23","arxiv_id":"1509.06928","code_links":[{"title":"Qatar-Computing-Research-Institute/dialectID","url":"https://github.com/Qatar-Computing-Research-Institute/dialectID"}],"syntology":null}},{"leaderboard":"/sota/spoken-language-identification-on-voxforge","slug":"spoken-language-identification-on-voxforge","dataset":"VoxForge Commonwealth","dataset_url":"/dataset/voxforge","rows_in_archive":4,"metrics":["Accuracy (%)"],"first_row_in_archive_order":{"model":"2D ConvNet(MixUp=YES)","paper_title":"Spoken Language Identification using ConvNets","paper_url":"/paper/spoken-language-identification-using-convnets","paper_date":"2019-10-09","arxiv_id":"1910.04269","code_links":[],"syntology":null}},{"leaderboard":"/sota/spoken-language-identification-on-indictts","slug":"spoken-language-identification-on-indictts","dataset":"IndicTTS","dataset_url":"/dataset/indictts","rows_in_archive":3,"metrics":["Classification Accuracy"],"first_row_in_archive_order":{"model":"CRNN","paper_title":"Is Attention always needed? A Case Study on Language Identification from Speech","paper_url":"/paper/is-attention-always-needed-a-case-study-on","paper_date":"2021-10-05","arxiv_id":"2110.03427","code_links":[],"syntology":null}},{"leaderboard":"/sota/spoken-language-identification-on-voxforge-2","slug":"spoken-language-identification-on-voxforge-2","dataset":"VoxForge","dataset_url":"/dataset/voxforge","rows_in_archive":3,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"LEAF","paper_title":"EfficientLEAF: A Faster LEarnable Audio Frontend of Questionable Use","paper_url":"/paper/efficientleaf-a-faster-learnable-audio","paper_date":"2022-07-12","arxiv_id":"2207.05508","code_links":[{"title":"cpjku/efficientleaf","url":"https://github.com/cpjku/efficientleaf"}],"syntology":null}},{"leaderboard":"/sota/spoken-language-identification-on","slug":"spoken-language-identification-on","dataset":"VOXLINGUA107","dataset_url":"/dataset/voxlingua107","rows_in_archive":2,"metrics":["0..5sec","5..20sec","Average"],"first_row_in_archive_order":{"model":"Noisy","paper_title":"VOXLINGUA107: A DATASET FOR SPOKEN LANGUAGE RECOGNITION","paper_url":"/paper/voxlingua107-a-dataset-for-spoken-language","paper_date":"2020-11-25","arxiv_id":"2011.12998","code_links":[],"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}}},{"leaderboard":"/sota/spoken-language-identification-on-kalaka-3","slug":"spoken-language-identification-on-kalaka-3","dataset":"KALAKA-3","dataset_url":null,"rows_in_archive":2,"metrics":["PC","EC","EO","PO"],"first_row_in_archive_order":{"model":"Model on the automatically filtered (cleaned) data","paper_title":"VOXLINGUA107: A DATASET FOR SPOKEN LANGUAGE RECOGNITION","paper_url":"/paper/voxlingua107-a-dataset-for-spoken-language","paper_date":"2020-11-25","arxiv_id":"2011.12998","code_links":[],"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}}},{"leaderboard":"/sota/spoken-language-identification-on-youtube-2","slug":"spoken-language-identification-on-youtube-2","dataset":"YouTube News dataset (Crackling Noise)","dataset_url":null,"rows_in_archive":2,"metrics":["Accuracy ","F1 Score"],"first_row_in_archive_order":{"model":"Inception-v3 CRNN","paper_title":"Language Identification Using Deep Convolutional Recurrent Neural Networks","paper_url":"/paper/language-identification-using-deep","paper_date":"2017-08-16","arxiv_id":"1708.04811","code_links":[{"title":"HPI-DeepLearning/crnn-lid","url":"https://github.com/HPI-DeepLearning/crnn-lid"}],"syntology":null}},{"leaderboard":"/sota/spoken-language-identification-on-youtube-3","slug":"spoken-language-identification-on-youtube-3","dataset":"YouTube News dataset (Background Music)","dataset_url":null,"rows_in_archive":2,"metrics":["Accuracy ","F1 Score"],"first_row_in_archive_order":{"model":"Inception-v3 CRNN","paper_title":"Language Identification Using Deep Convolutional Recurrent Neural Networks","paper_url":"/paper/language-identification-using-deep","paper_date":"2017-08-16","arxiv_id":"1708.04811","code_links":[{"title":"HPI-DeepLearning/crnn-lid","url":"https://github.com/HPI-DeepLearning/crnn-lid"}],"syntology":null}}],"datasets":[{"url":"/dataset/fleurs","name":"FLEURS","full_name":"Few-shot Learning Evaluation of Universal Representations of Speech","num_papers_in_archive":141},{"url":"/dataset/voxforge","name":"VoxForge","full_name":"VoxForge","num_papers_in_archive":11},{"url":"/dataset/indictts","name":"IndicTTS","full_name":"","num_papers_in_archive":3},{"url":"/dataset/voxlingua107","name":"VOXLINGUA107","full_name":"","num_papers_in_archive":2}],"subtasks":[],"parent_tasks":[{"url":"/task/spoken-language-understanding","name":"Spoken Language Understanding"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":13,"of":13,"tagged_in_all":51,"items":[{"url":"/paper/voxlingua107-a-dataset-for-spoken-language-1","title":"VoxLingua107: a Dataset for Spoken Language Recognition","date":"2020-11-25","arxiv_id":"2011.12998","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/afrihubert-a-self-supervised-speech","title":"AfriHuBERT: A self-supervised speech representation model for African languages","date":"2024-09-30","arxiv_id":"2409.20201","repositories_listed":1,"syntology":null},{"url":"/paper/improving-multilingual-asr-in-the-wild-using","title":"Improving Multilingual ASR in the Wild Using Simple N-best Re-ranking","date":"2024-09-27","arxiv_id":"2409.18428","repositories_listed":1,"syntology":null},{"url":"/paper/towards-training-bilingual-and-code-switched","title":"Unified model for code-switching speech recognition and language identification based on a concatenated tokenizer","date":"2023-06-14","arxiv_id":"2306.08753","repositories_listed":1,"syntology":null},{"url":"/paper/spoken-language-identification-system-for","title":"Spoken Language Identification System for English-Mandarin Code-Switching Child-Directed Speech","date":"2023-06-01","arxiv_id":"2306.00736","repositories_listed":1,"syntology":null},{"url":"/paper/improving-spoken-language-identification-with","title":"Improving Spoken Language Identification with Map-Mix","date":"2023-02-16","arxiv_id":"2302.08229","repositories_listed":1,"syntology":null},{"url":"/paper/efficientleaf-a-faster-learnable-audio","title":"EfficientLEAF: A Faster LEarnable Audio Frontend of Questionable Use","date":"2022-07-12","arxiv_id":"2207.05508","repositories_listed":1,"syntology":null},{"url":"/paper/distilled-non-semantic-speech-embeddings-with","title":"Distilled Non-Semantic Speech Embeddings with Binary Neural Networks for Low-Resource Devices","date":"2022-07-12","arxiv_id":"2207.05784","repositories_listed":1,"syntology":null},{"url":"/paper/bert-lid-leveraging-bert-to-improve-spoken","title":"BERT-LID: Leveraging BERT to Improve Spoken Language Identification","date":"2022-03-01","arxiv_id":"2203.00328","repositories_listed":1,"syntology":null},{"url":"/paper/triplet-entropy-loss-improving-the","title":"Triplet Entropy Loss: Improving The Generalisation of Short Speech Language Identification Systems","date":"2020-12-03","arxiv_id":"2012.03775","repositories_listed":1,"syntology":null},{"url":"/paper/cross-domain-adaptation-of-spoken-language","title":"Cross-Domain Adaptation of Spoken Language Identification for Related Languages: The Curious Case of Slavic Languages","date":"2020-08-02","arxiv_id":"2008.00545","repositories_listed":1,"syntology":null},{"url":"/paper/language-identification-using-deep","title":"Language Identification Using Deep Convolutional Recurrent Neural Networks","date":"2017-08-16","arxiv_id":"1708.04811","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-dialect-detection-in-arabic","title":"Automatic Dialect Detection in Arabic Broadcast Speech","date":"2015-09-23","arxiv_id":"1509.06928","repositories_listed":1,"syntology":null}],"syntology_records":1,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}