{"url":"/dataset/ravdess","name":"RAVDESS","full_name":"Ryerson Audio-Visual Database of Emotional Speech and Song","description_markdown":"The Ryerson Audio-Visual Database of Emotional Speech and Song (RAVDESS) contains 7,356 files (total size: 24.8 GB). The database contains 24 professional actors (12 female, 12 male), vocalizing two lexically-matched statements in a neutral North American accent. Speech includes calm, happy, sad, angry, fearful, surprise, and disgust expressions, and song contains calm, happy, sad, angry, and fearful emotions. Each expression is produced at two levels of emotional intensity (normal, strong), with an additional neutral expression. All conditions are available in three modality formats: Audio-only (16bit, 48kHz .wav), Audio-Video (720p H.264, AAC 48kHz, .mp4), and Video-only (no sound).  Note, there are no song files for Actor_18.\r\n\r\nPaper: [The Ryerson Audio-Visual Database of Emotional Speech and Song (RAVDESS): A dynamic, multimodal set of facial and vocal expressions in North American English](https://doi.org/10.1371/journal.pone.0196391)\r\nSource: [](https://zenodo.org/record/1188976#.YFZuJ0j7SL8)","description_withheld":null,"homepage":"https://zenodo.org/record/1188976#.YFZuJ0j7SL8","introduced_date":"2020-12-31","introduced_date_note":null,"introduced_by":null,"license":{"name":"Attribution-NonCommercial-ShareAlike 4.0 International","url":"https://creativecommons.org/licenses/by-nc-sa/4.0/legalcode"},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Audio","url":"/datasets/modality/audio"},{"name":"Speech","url":"/datasets/modality/speech"}],"tasks":[{"name":"Emotion Recognition","url":"/task/emotion-recognition","datasets_with_task":"/datasets/task/emotion-recognition"},{"name":"Facial Expression Recognition (FER)","url":"/task/facial-expression-recognition","datasets_with_task":"/datasets/task/facial-expression-recognition"},{"name":"Audio Classification","url":"/task/audio-classification","datasets_with_task":"/datasets/task/audio-classification"},{"name":"Speech Emotion Recognition","url":"/task/speech-emotion-recognition","datasets_with_task":"/datasets/task/speech-emotion-recognition"},{"name":"Emotion Classification","url":"/task/emotion-classification","datasets_with_task":"/datasets/task/emotion-classification"},{"name":"Video Emotion Recognition","url":"/task/video-emotion-recognition","datasets_with_task":"/datasets/task/video-emotion-recognition"},{"name":"Facial Emotion Recognition","url":"/task/facial-emotion-recognition","datasets_with_task":"/datasets/task/facial-emotion-recognition"},{"name":"Music Emotion Recognition","url":"/task/music-emotion-recognition","datasets_with_task":"/datasets/task/music-emotion-recognition"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["RAVDESS"],"data_loaders":[{"repo":"https://github.com/activeloopai/Hub","url":"https://docs.activeloop.ai/datasets/ravdess-dataset","frameworks":["tf","pytorch"]}],"num_papers_in_archive":27,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/emotion-recognition-on-ravdess","task":"Emotion Recognition","dataset_variant":"RAVDESS","rows":5,"metrics":["Accuracy","WAR"],"first_row_in_archive_order":{"model":"LogisticRegression on posteriors of xlsr-Wav2Vec2.0&bi-LSTM+Attention","paper":"/paper/a-proposal-for-multimodal-emotion-recognition","metrics":{"Accuracy":"86.70%"},"code_links":[{"title":"cristinalunaj/MMEmotionRecognition","url":"https://github.com/cristinalunaj/MMEmotionRecognition"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speech-emotion-recognition-on-ravdess","task":"Speech Emotion Recognition","dataset_variant":"RAVDESS","rows":5,"metrics":["Accuracy","F1 Score","Precision","Recall","F1"],"first_row_in_archive_order":{"model":"VQ-MAE-S-12 (Frame) + Query2Emo","paper":"/paper/a-vector-quantized-masked-autoencoder-for","metrics":{"Accuracy":"84.1","F1":"0.844"},"code_links":[{"title":"samsad35/VQ-MAE-S-code","url":"https://github.com/samsad35/VQ-MAE-S-code"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/facial-emotion-recognition-on-ravdess","task":"Facial Emotion Recognition","dataset_variant":"RAVDESS","rows":4,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"MTCAE-DFER","paper":"/paper/mtcae-dfer-multi-task-cascaded-autoencoder","metrics":{"Accuracy":"83.69%"},"code_links":[{"title":"Peihao-Xiang/MTCAE-DFER","url":"https://github.com/Peihao-Xiang/MTCAE-DFER"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/audio-classification-on-ravdess","task":"Audio Classification","dataset_variant":"RAVDESS","rows":2,"metrics":["Top-1 Accuracy"],"first_row_in_archive_order":{"model":"ASM-RH-A","paper":"/paper/mixer-is-more-than-just-a-model","metrics":{"Top-1 Accuracy":"75.4"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/emotion-classification-on-ravdess","task":"Emotion Classification","dataset_variant":"RAVDESS","rows":1,"metrics":["Top-1 Accuracy"],"first_row_in_archive_order":{"model":"ERANN-0-4","paper":null,"metrics":{"Top-1 Accuracy":"74.8"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/facial-expression-recognition-on-ravdess","task":"Facial Expression Recognition (FER)","dataset_variant":"RAVDESS","rows":1,"metrics":["UAR"],"first_row_in_archive_order":{"model":"EmoAffectNet LSTM","paper":"/paper/in-search-of-a-robust-facial-expressions","metrics":{"UAR":"69.7"},"code_links":[{"title":"ElenaRyumina/EMO-AffectNetModel","url":"https://github.com/ElenaRyumina/EMO-AffectNetModel"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/mtcae-dfer-multi-task-cascaded-autoencoder","title":"MTCAE-DFER: Multi-Task Cascaded Autoencoder for Dynamic Facial Expression Recognition","date":"2024-12-25","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/multimae-der-multimodal-masked-autoencoder","title":"MultiMAE-DER: Multimodal Masked Autoencoder for Dynamic Emotion Recognition","date":"2024-04-28","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/mixer-is-more-than-just-a-model","title":"Mixer is more than just a model","date":"2024-02-28","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/a-vector-quantized-masked-autoencoder-for","title":"A vector quantized masked autoencoder for speech emotion recognition","date":"2023-04-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/in-search-of-a-robust-facial-expressions","title":"In Search of a Robust Facial Expressions Recognition Model: A Large-Scale Visual Cross-Corpus Study","date":"2022-10-07","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/self-attention-fusion-for-audiovisual-emotion","title":"Self-attention fusion for audiovisual emotion recognition with incomplete data","date":"2022-01-26","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":2,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/a-proposal-for-multimodal-emotion-recognition","title":"A proposal for Multimodal Emotion Recognition using aural transformers and Action Units on RAVDESS dataset","date":"2021-12-30","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/multimodal-emotion-recognition-on-ravdess","title":"Multimodal Emotion Recognition on RAVDESS Dataset Using Transfer Learning","date":"2021-11-18","rows_on_this_dataset":4,"code_links":0,"syntology":null},{"paper":"/paper/shallow-over-deep-neural-networks-a-empirical","title":"Shallow over Deep Neural Networks: A empirical analysis for human emotion classification using audio data","date":"2020-07-03","rows_on_this_dataset":1,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":8,"samples_ran":2,"samples_unverified":6,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}