{"url":"/dataset/cmu-mosei","name":"CMU-MOSEI","full_name":null,"description_markdown":"CMU Multimodal Opinion Sentiment and Emotion Intensity (**CMU-MOSEI**) is the largest dataset of sentence-level sentiment analysis and emotion recognition in online videos. CMU-MOSEI contains over 12 hours of annotated video from over 1000 speakers and 250 topics.","description_withheld":null,"homepage":"","introduced_date":"2018-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/multimodal-language-analysis-in-the-wild-cmu","title":"Multimodal Language Analysis in the Wild: CMU-MOSEI Dataset and Interpretable Dynamic Fusion Graph","first_author":"AmirAli Bagher Zadeh","url":null},"license":{"name":"Custom","url":"https://github.com/A2Zadeh/CMU-MultimodalSDK/blob/master/LICENSE.txt"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Texts","url":"/datasets/modality/texts"},{"name":"Audio","url":"/datasets/modality/audio"}],"tasks":[{"name":"Facial Expression Recognition","url":"/task/facial-expression-recognition-1","datasets_with_task":"/datasets/task/facial-expression-recognition-1"},{"name":"Emotion Classification","url":"/task/emotion-classification","datasets_with_task":"/datasets/task/emotion-classification"},{"name":"Multimodal Emotion Recognition","url":"/task/multimodal-emotion-recognition","datasets_with_task":"/datasets/task/multimodal-emotion-recognition"},{"name":"Multimodal Sentiment Analysis","url":"/task/multimodal-sentiment-analysis","datasets_with_task":"/datasets/task/multimodal-sentiment-analysis"},{"name":"Video Emotion Detection","url":"/task/video-emotion-detection","datasets_with_task":"/datasets/task/video-emotion-detection"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["CMU-MOSEI"],"data_loaders":[{"repo":"https://github.com/lobracost/MultimodalSDK_loader","url":"https://github.com/lobracost/MultimodalSDK_loader","frameworks":[]}],"num_papers_in_archive":190,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/multimodal-sentiment-analysis-on-cmu-mosei-1","task":"Multimodal Sentiment Analysis","dataset_variant":"CMU-MOSEI","rows":15,"metrics":["Accuracy","MAE","F1","Acc-5","Acc-7","Corr"],"first_row_in_archive_order":{"model":"SeMUL-PCD","paper":"/paper/multi-label-emotion-analysis-in-conversation","metrics":{"Accuracy":"88.62","F1":"89.04"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/emotion-classification-on-cmu-mosei","task":"Emotion Classification","dataset_variant":"CMU-MOSEI","rows":4,"metrics":["Accuracy","Weighted Accuracy"],"first_row_in_archive_order":{"model":"MARLIN (ViT-L)","paper":"/paper/marlin-masked-autoencoder-for-facial-video","metrics":{"Accuracy":"80.63"},"code_links":[{"title":"ControlNet/MARLIN","url":"https://github.com/ControlNet/MARLIN"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/facial-expression-recognition-on-cmu-mosei","task":"Facial Expression Recognition","dataset_variant":"CMU-MOSEI","rows":1,"metrics":["Weighted Accuracy"],"first_row_in_archive_order":{"model":"ConCluGen","paper":"/paper/multi-task-multi-modal-self-supervised","metrics":{"Weighted Accuracy":"66.48"},"code_links":[{"title":"tub-cv-group/conclugen","url":"https://github.com/tub-cv-group/conclugen"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/multi-task-multi-modal-self-supervised","title":"Multi-Task Multi-Modal Self-Supervised Learning for Facial Expression Recognition","date":"2024-04-16","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/multi-label-emotion-analysis-in-conversation","title":"Multi-label Emotion Analysis in Conversation via Multimodal Knowledge Distillation","date":"2023-10-27","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/learning-language-guided-adaptive-hyper","title":"Learning Language-guided Adaptive Hyper-modality Representation for Multimodal Sentiment Analysis","date":"2023-10-09","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/multi-modality-multi-loss-fusion-network","title":"Multimodal Multi-loss Fusion Network for Sentiment Analysis","date":"2023-08-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/speech-text-dialog-pre-training-for-spoken","title":"Speech-Text Dialog Pre-training for Spoken Dialog Understanding with Explicit Cross-Modal Alignment","date":"2023-05-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/unimse-towards-unified-multimodal-sentiment","title":"UniMSE: Towards Unified Multimodal Sentiment Analysis and Emotion Recognition","date":"2022-11-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/marlin-masked-autoencoder-for-facial-video","title":"MARLIN: Masked Autoencoder for facial video Representation LearnINg","date":"2022-11-12","rows_on_this_dataset":6,"code_links":1,"syntology":null},{"paper":"/paper/mmlatch-bottom-up-top-down-fusion-for","title":"MMLatch: Bottom-up Top-down Fusion for Multimodal Sentiment Analysis","date":"2022-01-24","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/unsupervised-multimodal-language","title":"Unsupervised Multimodal Language Representations using Convolutional Autoencoders","date":"2021-10-06","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/modulated-fusion-using-transformer-for","title":"Modulated Fusion using Transformer for Linguistic-Acoustic Emotion Recognition","date":"2020-10-05","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/a-transformer-based-joint-encoding-for-1","title":"A Transformer-based joint-encoding for Emotion Recognition and Sentiment Analysis","date":"2020-06-29","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":1,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/gated-mechanism-for-attention-based","title":"Gated Mechanism for Attention Based Multimodal Sentiment Analysis","date":"2020-02-21","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/multilogue-net-a-context-aware-rnn-for-multi","title":"Multilogue-Net: A Context Aware RNN for Multi-modal Emotion Detection and Sentiment Analysis in Conversation","date":"2020-02-19","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":2,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multimodal-language-analysis-in-the-wild-cmu","title":"Multimodal Language Analysis in the Wild: CMU-MOSEI Dataset and Interpretable Dynamic Fusion Graph","date":"2018-07-01","rows_on_this_dataset":1,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":2,"samples_harvested":8,"samples_ran":3,"samples_unverified":5,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}