{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/massive-a-1m-example-multilingual-natural","title":"MASSIVE: A 1M-Example Multilingual Natural Language Understanding Dataset with 51 Typologically-Diverse Languages","arxiv_id":"2204.08582","date":"2022-04-18","proceeding":null,"authors":["Jack FitzGerald","Christopher Hench","Charith Peris","Scott Mackie","Kay Rottmann","Ana Sanchez","Aaron Nash","Liam Urbach","Vishesh Kakarala","Richa Singh","Swetha Ranganath","Laurie Crist","Misha Britan","Wouter Leeuwis","Gokhan Tur","Prem Natarajan"],"abstract":"We present the MASSIVE dataset--Multilingual Amazon Slu resource package (SLURP) for Slot-filling, Intent classification, and Virtual assistant Evaluation. MASSIVE contains 1M realistic, parallel, labeled virtual assistant utterances spanning 51 languages, 18 domains, 60 intents, and 55 slots. MASSIVE was created by tasking professional translators to localize the English-only SLURP dataset into 50 typologically diverse languages from 29 genera. We also present modeling results on XLM-R and mT5, including exact match accuracy, intent classification accuracy, and slot-filling F1 score. We have released our dataset, modeling code, and models publicly.","url_abs":"https://arxiv.org/abs/2204.08582v2","url_pdf":"https://arxiv.org/pdf/2204.08582v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"massive-a-1m-example-multilingual-natural","repo_url":"https://github.com/alexa/massive","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"massive-a-1m-example-multilingual-natural","repo_url":"https://bitbucket.org/robvanderg/sid4lr","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":null},{"paper_slug":"massive-a-1m-example-multilingual-natural","repo_url":"https://github.com/ai4bharat/indicbert","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"massive-a-1m-example-multilingual-natural","repo_url":"https://github.com/hlt-mt/speech-massive","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"Apache-2.0"}},{"paper_slug":"massive-a-1m-example-multilingual-natural","repo_url":"https://github.com/pswietojanski/slurp","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok","spdx":"NOASSERTION"}},{"paper_slug":"massive-a-1m-example-multilingual-natural","repo_url":"https://github.com/rita-nlp/italic","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"Apache-2.0"}}],"tasks":[{"task_slug":"intent-classification","task_name":"Intent Classification"},{"task_slug":"natural-language-understanding","task_name":"Natural Language Understanding"},{"task_slug":"slot-filling","task_name":"Slot Filling"},{"task_slug":"xlm-r","task_name":"XLM-R"},{"task_slug":"zero-shot-intent-classification","task_name":"Zero-Shot Intent Classification"},{"task_slug":"zero-shot-slot-filling","task_name":"Zero-shot Slot Filling"},{"task_slug":"intent-classification-1","task_name":"intent-classification"}],"methods":[{"method_slug":"adafactor","method_name":"Adafactor"},{"method_slug":"attention","method_name":"Attention"},{"method_slug":"attention-dropout","method_name":"Attention Dropout"},{"method_slug":"bpe","method_name":"BPE"},{"method_slug":"dense-connections","method_name":"Dense Connections"},{"method_slug":"dropout","method_name":"Dropout"},{"method_slug":"glu","method_name":"Gated Linear Unit"},{"method_slug":"inverse-square-root-schedule","method_name":"Inverse Square Root Schedule"},{"method_slug":"layer-normalization","method_name":"Layer Normalization"},{"method_slug":"linear-layer","method_name":"Linear Layer"},{"method_slug":"multi-head-attention","method_name":"Multi-Head Attention"},{"method_slug":"residual-connection","method_name":"Residual Connection"},{"method_slug":"sentencepiece","method_name":"SentencePiece"},{"method_slug":"softmax","method_name":"Softmax"},{"method_slug":"t5","method_name":"T5"},{"method_slug":"xlm-r","method_name":"XLM-R"},{"method_slug":"mt5","method_name":"mT5"}],"datasets_introduced":[{"slug":"massive","name":"MASSIVE","full_name":""}],"methods_introduced":[],"results":[{"leaderboard":"/sota/intent-classification-on-massive","task":"Intent Classification","dataset":"MASSIVE","model":"mT5 Base (encoder-only)","rank_in_archive_order":1,"of":3,"metrics":{"Intent Accuracy":"86.1"},"uses_additional_data":false},{"leaderboard":"/sota/intent-classification-on-massive","task":"Intent Classification","dataset":"MASSIVE","model":"mT5 Base (text-to-text)","rank_in_archive_order":2,"of":3,"metrics":{"Intent Accuracy":"85.3"},"uses_additional_data":false},{"leaderboard":"/sota/intent-classification-on-massive","task":"Intent Classification","dataset":"MASSIVE","model":"XLM-R Base","rank_in_archive_order":3,"of":3,"metrics":{"Intent Accuracy":"85.1"},"uses_additional_data":false},{"leaderboard":"/sota/slot-filling-on-massive","task":"Slot Filling","dataset":"MASSIVE","model":"XLM-R Base","rank_in_archive_order":1,"of":3,"metrics":{"Slot F1 Score":"83.6"},"uses_additional_data":false},{"leaderboard":"/sota/slot-filling-on-massive","task":"Slot Filling","dataset":"MASSIVE","model":"mT5 Base (encoder-only)","rank_in_archive_order":2,"of":3,"metrics":{"Slot F1 Score":"82.2"},"uses_additional_data":false},{"leaderboard":"/sota/slot-filling-on-massive","task":"Slot Filling","dataset":"MASSIVE","model":"mT5 Base (text-to-text)","rank_in_archive_order":3,"of":3,"metrics":{"Slot F1 Score":"81.3"},"uses_additional_data":false},{"leaderboard":"/sota/zero-shot-slot-filling-on-massive","task":"Zero-shot Slot Filling","dataset":"MASSIVE","model":"XLM-R Base","rank_in_archive_order":1,"of":3,"metrics":{"Slot F1 Score":"64.2"},"uses_additional_data":false},{"leaderboard":"/sota/zero-shot-slot-filling-on-massive","task":"Zero-shot Slot Filling","dataset":"MASSIVE","model":"mT5 Base (encoder-only)","rank_in_archive_order":2,"of":3,"metrics":{"Slot F1 Score":"56.9"},"uses_additional_data":false},{"leaderboard":"/sota/zero-shot-slot-filling-on-massive","task":"Zero-shot Slot Filling","dataset":"MASSIVE","model":"mT5 Base (text-to-text)","rank_in_archive_order":3,"of":3,"metrics":{"Slot F1 Score":"50.6"},"uses_additional_data":false}],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2204.08582","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.08582"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/hlt-mt/speech-massive","reach":{"status":"ok","spdx":"Apache-2.0"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://bitbucket.org/robvanderg/sid4lr","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/alexa/massive","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/pswietojanski/slurp","reach":{"status":"ok","spdx":"NOASSERTION"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/ai4bharat/indicbert","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/rita-nlp/italic","reach":{"status":"ok","spdx":"Apache-2.0"}}],"summary":{"ran_violates":1,"unverified":9},"by_repo_kind":{"official":{"samples":1,"ran":1,"repositories":1},"listed":{"samples":9,"ran":0,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":1,"samples":[{"code_sha256_prefix":"00de6205350da403","entry":"isascii","repo":"alexa/massive","repo_kind":"official","path":"scripts/create_hf_dataset.py","file_url":"https://github.com/alexa/massive/blob/HEAD/scripts/create_hf_dataset.py","link_basis":"first_harvest_node","language":"python","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"mcp_get_code":{"code_sha256":"00de6205350da403"}},{"code_sha256_prefix":"652b541e7a5feac9","entry":"compute_metrics","repo":"rita-nlp/italic","repo_kind":"listed","path":"text_finetuning.py","file_url":"https://github.com/rita-nlp/italic/blob/HEAD/text_finetuning.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"652b541e7a5feac9"}},{"code_sha256_prefix":"0a2872c4352696b2","entry":"compute_metrics","repo":"rita-nlp/italic","repo_kind":"listed","path":"text_inference.py","file_url":"https://github.com/rita-nlp/italic/blob/HEAD/text_inference.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"0a2872c4352696b2"}},{"code_sha256_prefix":"f3b2d10b53645633","entry":"compute_metrics","repo":"rita-nlp/italic","repo_kind":"listed","path":"utils.py","file_url":"https://github.com/rita-nlp/italic/blob/HEAD/utils.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"f3b2d10b53645633"}},{"code_sha256_prefix":"53d7e4178b94a91c","entry":"define_model","repo":"rita-nlp/italic","repo_kind":"listed","path":"ic_finetuning.py","file_url":"https://github.com/rita-nlp/italic/blob/HEAD/ic_finetuning.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"53d7e4178b94a91c"}},{"code_sha256_prefix":"d253d61bd1e78474","entry":"define_model","repo":"rita-nlp/italic","repo_kind":"listed","path":"ic_inference.py","file_url":"https://github.com/rita-nlp/italic/blob/HEAD/ic_inference.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"d253d61bd1e78474"}},{"code_sha256_prefix":"fef1c66dfeeb6664","entry":"get_text","repo":"rita-nlp/italic","repo_kind":"listed","path":"asr_inference.py","file_url":"https://github.com/rita-nlp/italic/blob/HEAD/asr_inference.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"fef1c66dfeeb6664"}},{"code_sha256_prefix":"0a219dbdc5c3d3aa","entry":"is_audio_in_length_range","repo":"rita-nlp/italic","repo_kind":"listed","path":"asr_finetuning.py","file_url":"https://github.com/rita-nlp/italic/blob/HEAD/asr_finetuning.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"0a219dbdc5c3d3aa"}},{"code_sha256_prefix":"b1dc84afdc06af86","entry":"is_target_text_in_range","repo":"rita-nlp/italic","repo_kind":"listed","path":"asr_inference.py","file_url":"https://github.com/rita-nlp/italic/blob/HEAD/asr_inference.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"b1dc84afdc06af86"}},{"code_sha256_prefix":"48422a4b6a62ca2f","entry":"normalise","repo":"rita-nlp/italic","repo_kind":"listed","path":"asr_inference.py","file_url":"https://github.com/rita-nlp/italic/blob/HEAD/asr_inference.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"48422a4b6a62ca2f"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}