{"url":"/dataset/massive","name":"MASSIVE","full_name":null,"description_markdown":"MASSIVE is a parallel dataset of > 1M utterances across 51 languages with annotations for the Natural Language Understanding tasks of intent prediction and slot annotation. Utterances span 60 intents and include 55 slot types. MASSIVE was created by localizing the SLURP dataset, composed of general Intelligent Voice Assistant single-shot interactions.","description_withheld":null,"homepage":"https://github.com/alexa/massive","introduced_date":"2022-04-18","introduced_date_note":null,"introduced_by":{"paper":"/paper/massive-a-1m-example-multilingual-natural","title":"MASSIVE: A 1M-Example Multilingual Natural Language Understanding Dataset with 51 Typologically-Diverse Languages","first_author":"Jack FitzGerald","url":null},"license":{"name":"CC BY 4.0","url":"https://github.com/alexa/massive/blob/main/NOTICE.md#notice"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Slot Filling","url":"/task/slot-filling","datasets_with_task":"/datasets/task/slot-filling"},{"name":"Intent Classification","url":"/task/intent-classification","datasets_with_task":"/datasets/task/intent-classification"},{"name":"Zero-shot Slot Filling","url":"/task/zero-shot-slot-filling","datasets_with_task":"/datasets/task/zero-shot-slot-filling"},{"name":"Intent Classification and Slot Filling","url":"/task/intent-classification-and-slot-filling","datasets_with_task":"/datasets/task/intent-classification-and-slot-filling"},{"name":"Zero-Shot Intent Classification","url":"/task/zero-shot-intent-classification","datasets_with_task":"/datasets/task/zero-shot-intent-classification"},{"name":"Zero-Shot Intent Classification and Slot Filling","url":"/task/zero-shot-intent-classification-and-slot","datasets_with_task":"/datasets/task/zero-shot-intent-classification-and-slot"}],"languages":[],"variants":["MASSIVE"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/nouamanetazi/test123","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/AmazonScience/massive","frameworks":["tf","pytorch","jax"]}],"num_papers_in_archive":72,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/intent-classification-on-massive","task":"Intent Classification","dataset_variant":"MASSIVE","rows":3,"metrics":["Intent Accuracy"],"first_row_in_archive_order":{"model":"mT5 Base (encoder-only)","paper":"/paper/massive-a-1m-example-multilingual-natural","metrics":{"Intent Accuracy":"86.1"},"code_links":[{"title":"alexa/massive","url":"https://github.com/alexa/massive"},{"title":"pswietojanski/slurp","url":"https://github.com/pswietojanski/slurp"},{"title":"ai4bharat/indicbert","url":"https://github.com/ai4bharat/indicbert"},{"title":"hlt-mt/speech-massive","url":"https://github.com/hlt-mt/speech-massive"},{"title":"rita-nlp/italic","url":"https://github.com/rita-nlp/italic"},{"title":"robvanderg/sid4lr","url":"https://bitbucket.org/robvanderg/sid4lr"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/slot-filling-on-massive","task":"Slot Filling","dataset_variant":"MASSIVE","rows":3,"metrics":["Slot F1 Score"],"first_row_in_archive_order":{"model":"XLM-R Base","paper":"/paper/massive-a-1m-example-multilingual-natural","metrics":{"Slot F1 Score":"83.6"},"code_links":[{"title":"alexa/massive","url":"https://github.com/alexa/massive"},{"title":"pswietojanski/slurp","url":"https://github.com/pswietojanski/slurp"},{"title":"ai4bharat/indicbert","url":"https://github.com/ai4bharat/indicbert"},{"title":"hlt-mt/speech-massive","url":"https://github.com/hlt-mt/speech-massive"},{"title":"rita-nlp/italic","url":"https://github.com/rita-nlp/italic"},{"title":"robvanderg/sid4lr","url":"https://bitbucket.org/robvanderg/sid4lr"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zero-shot-slot-filling-on-massive","task":"Zero-shot Slot Filling","dataset_variant":"MASSIVE","rows":3,"metrics":["Slot F1 Score"],"first_row_in_archive_order":{"model":"XLM-R Base","paper":"/paper/massive-a-1m-example-multilingual-natural","metrics":{"Slot F1 Score":"64.2"},"code_links":[{"title":"alexa/massive","url":"https://github.com/alexa/massive"},{"title":"pswietojanski/slurp","url":"https://github.com/pswietojanski/slurp"},{"title":"ai4bharat/indicbert","url":"https://github.com/ai4bharat/indicbert"},{"title":"hlt-mt/speech-massive","url":"https://github.com/hlt-mt/speech-massive"},{"title":"rita-nlp/italic","url":"https://github.com/rita-nlp/italic"},{"title":"robvanderg/sid4lr","url":"https://bitbucket.org/robvanderg/sid4lr"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/massive-a-1m-example-multilingual-natural","title":"MASSIVE: A 1M-Example Multilingual Natural Language Understanding Dataset with 51 Typologically-Diverse Languages","date":"2022-04-18","rows_on_this_dataset":9,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":1,"samples_unverified":9,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":10,"samples_ran":1,"samples_unverified":9,"pointer_only_for_licence":1,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}