{"url":"/dataset/mass","name":"MaSS","full_name":null,"description_markdown":"MaSS (Multilingual corpus of Sentence-aligned Spoken utterances) is an extension of the CMU Wilderness Multilingual Speech Dataset, a speech dataset based on recorded readings of the New Testament.\r\n\r\nMaSS extends it by providing a large and clean dataset of 8,130 parallel spoken utterances across 8 languages (56 language pairs). The covered languages are: Basque, English, Finnish, French, Hungarian, Romanian, Russian and Spanish.\r\n\r\nSource: [MaSS: A Large and Clean Multilingual Corpus of Sentence-aligned Spoken Utterances Extracted from the Bible](/paper/mass-a-large-and-clean-multilingual-corpus-of)\r\nImage Source: [https://arxiv.org/pdf/1907.12895v3.pdf](https://arxiv.org/pdf/1907.12895v3.pdf)","description_withheld":null,"homepage":"https://github.com/getalp/mass-dataset","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/mass-a-large-and-clean-multilingual-corpus-of","title":"MaSS: A Large and Clean Multilingual Corpus of Sentence-aligned Spoken Utterances Extracted from the Bible","first_author":"Marcely Zanon Boito","url":null},"license":null,"modalities":[{"name":"Speech","url":"/datasets/modality/speech"}],"tasks":[{"name":"Speech Recognition","url":"/task/speech-recognition","datasets_with_task":"/datasets/task/speech-recognition"},{"name":"Video Captioning","url":"/task/video-captioning","datasets_with_task":"/datasets/task/video-captioning"},{"name":"Sleep Stage Detection","url":"/task/sleep-stage-detection","datasets_with_task":"/datasets/task/sleep-stage-detection"},{"name":"Spindle Detection","url":"/task/spindle-detection","datasets_with_task":"/datasets/task/spindle-detection"},{"name":"K-complex detection","url":"/task/k-complex-detection","datasets_with_task":"/datasets/task/k-complex-detection"}],"languages":[{"name":"English","url":"/datasets/language/english"},{"name":"French","url":"/datasets/language/french"},{"name":"Spanish","url":"/datasets/language/spanish"},{"name":"Russian","url":"/datasets/language/russian"},{"name":"Basque","url":"/datasets/language/basque"},{"name":"Finnish","url":"/datasets/language/finnish"},{"name":"Hungarian","url":"/datasets/language/hungarian"},{"name":"Romanian","url":"/datasets/language/romanian"}],"variants":["MASS SS2","MaSS"],"data_loaders":[{"repo":"https://github.com/getalp/mass-dataset","url":"https://github.com/getalp/mass-dataset","frameworks":[]}],"num_papers_in_archive":15,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/spindle-detection-on-mass-ss2","task":"Spindle Detection","dataset_variant":"MASS SS2","rows":6,"metrics":["F1-score (@IoU = 0.3)","F1-score (@IoU = 0.2)"],"first_row_in_archive_order":{"model":"RED-Time","paper":"/paper/red-deep-recurrent-neural-networks-for-sleep","metrics":{"F1-score (@IoU = 0.3)":"0.812"},"code_links":[{"title":"nicolasigor/cmorlet-tensorflow","url":"https://github.com/nicolasigor/cmorlet-tensorflow"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/k-complex-detection-on-mass-ss2","task":"K-complex detection","dataset_variant":"MASS SS2","rows":4,"metrics":["F1-score (@IoU = 0.3)","F1-score (@IoU = 0.2)"],"first_row_in_archive_order":{"model":"RED-CWT","paper":"/paper/red-deep-recurrent-neural-networks-for-sleep","metrics":{"F1-score (@IoU = 0.2)":"0.826","F1-score (@IoU = 0.3)":"0.827"},"code_links":[{"title":"nicolasigor/cmorlet-tensorflow","url":"https://github.com/nicolasigor/cmorlet-tensorflow"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/sleep-stage-detection-on-mass-ss2","task":"Sleep Stage Detection","dataset_variant":"MASS SS2","rows":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"IITNet (F4-EOG [Left] only)","paper":"/paper/intra-and-inter-epoch-temporal-context","metrics":{"Accuracy":"84.5%"},"code_links":[{"title":"gist-ailab/IITNet-official","url":"https://github.com/gist-ailab/IITNet-official"},{"title":"SandraPrestel/deep-sleep-transfer","url":"https://github.com/SandraPrestel/deep-sleep-transfer"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/red-deep-recurrent-neural-networks-for-sleep","title":"RED: Deep Recurrent Neural Networks for Sleep EEG Event Detection","date":"2020-05-15","rows_on_this_dataset":4,"code_links":1,"syntology":null},{"paper":"/paper/intra-and-inter-epoch-temporal-context","title":"Intra- and Inter-epoch Temporal Context Network (IITNet) Using Sub-epoch Features for Automatic Sleep Scoring on Raw Single-channel EEG","date":"2019-02-18","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/dosed-a-deep-learning-approach-to-detect","title":"DOSED: a deep learning approach to detect multiple sleep micro-events in EEG signal","date":"2018-12-07","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/joint-classification-and-prediction-cnn","title":"Joint Classification and Prediction CNN Framework for Automatic Sleep Stage Classification","date":"2018-05-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/a-single-channel-sleep-spindle-detector-based","title":"A single channel sleep-spindle detector based on multivariate classification of EEG epochs: MUSSDET.","date":"2018-03-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/multichannel-sleep-spindle-detection-using","title":"Multichannel sleep spindle detection using sparse low-rank optimization","date":"2017-08-15","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/meet-spinky-an-open-source-spindle-and-k","title":"Meet Spinky: An Open-Source Spindle and K-Complex Detection Toolbox Validated on the Open-Access Montreal Archive of Sleep Studies (MASS).","date":"2017-03-02","rows_on_this_dataset":2,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}