{"url":"/dataset/iemocap","name":"IEMOCAP","full_name":"The Interactive Emotional Dyadic Motion Capture (IEMOCAP) Database","description_markdown":"Multimodal Emotion Recognition **IEMOCAP** The IEMOCAP dataset consists of 151 videos of recorded dialogues, with 2 speakers per session for a total of 302 videos across the dataset. Each segment is annotated for the presence of 9 emotions (angry, excited, fear, sad, surprised, frustrated, happy, disappointed and neutral) as well as valence, arousal and dominance. The dataset is recorded across 5 sessions with 5 pairs of speakers.\r\n\r\nSource: [Multi-attention Recurrent Network for Human Communication Comprehension](https://arxiv.org/abs/1802.00923)\r\nImage Source: [https://sail.usc.edu/iemocap/Busso_2008_iemocap.pdf](https://sail.usc.edu/iemocap/Busso_2008_iemocap.pdf)","description_withheld":null,"homepage":"https://sail.usc.edu/iemocap/iemocap_publication.htm","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":null,"title":"IEMOCAP: interactive emotional dyadic motion capture database","first_author":null,"url":"https://doi.org/10.1007/s10579-008-9076-6"},"license":{"name":"Custom (non-commercial)","url":"https://sail.usc.edu/iemocap/Data_Release_Form_IEMOCAP.pdf"},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Audio","url":"/datasets/modality/audio"}],"tasks":[{"name":"Speech Emotion Recognition","url":"/task/speech-emotion-recognition","datasets_with_task":"/datasets/task/speech-emotion-recognition"},{"name":"Emotion Recognition in Conversation","url":"/task/emotion-recognition-in-conversation","datasets_with_task":"/datasets/task/emotion-recognition-in-conversation"},{"name":"Multimodal Emotion Recognition","url":"/task/multimodal-emotion-recognition","datasets_with_task":"/datasets/task/multimodal-emotion-recognition"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["IEMOCAP"],"data_loaders":[{"repo":"https://github.com/macaixia84/dialogue-generation","url":"https://github.com/macaixia84/dialogue-generation","frameworks":["tf","pytorch"]}],"num_papers_in_archive":749,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/emotion-recognition-in-conversation-on","task":"Emotion Recognition in Conversation","dataset_variant":"IEMOCAP","rows":59,"metrics":["Weighted-F1","Accuracy","Micro-F1","Macro-F1"],"first_row_in_archive_order":{"model":"SDT","paper":"/paper/a-transformer-based-model-with-self","metrics":{"Accuracy":"73.95","Weighted-F1":"74.08"},"code_links":[{"title":"butterfliesss/sdt","url":"https://github.com/butterfliesss/sdt"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speech-emotion-recognition-on-iemocap","task":"Speech Emotion Recognition","dataset_variant":"IEMOCAP","rows":8,"metrics":["UA CV","WA CV","UA","WA","F1"],"first_row_in_archive_order":{"model":"SER with MTL","paper":"/paper/speech-emotion-recognition-with-multi-task","metrics":{"F1":"-","UA CV":"0.7815"},"code_links":[{"title":"TideDancer/interspeech21_emotion","url":"https://github.com/TideDancer/interspeech21_emotion"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/multimodal-emotion-recognition-on-iemocap","task":"Multimodal Emotion Recognition","dataset_variant":"IEMOCAP","rows":2,"metrics":["Weighted F1","Accuracy"],"first_row_in_archive_order":{"model":"GraphSmile","paper":"/paper/2407-21536","metrics":{"Accuracy":"72.77","Weighted F1":"72.81"},"code_links":[{"title":"lijfrank-open/GraphSmile","url":"https://github.com/lijfrank-open/GraphSmile"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/long-short-distance-graph-neural-networks-and","title":"Long-Short Distance Graph Neural Networks and Improved Curriculum Learning for Emotion Recognition in Conversation","date":"2025-07-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/2407-21536","title":"Tracing Intricate Cues in Dialogue: Joint Graph Structure and Sentiment Dynamics for Multimodal Emotion Recognition","date":"2024-07-31","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/2407-21315","title":"Beyond Silent Letters: Amplifying LLMs in Emotion Recognition with Vocal Nuances","date":"2024-07-31","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/bioserc-integrating-biography-speakers","title":"BiosERC: Integrating Biography Speakers Supported by LLMs for ERC Tasks","date":"2024-07-05","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/efficient-long-distance-latent-relation-aware","title":"Efficient Long-distance Latent Relation-aware Graph Neural Network for Multi-modal Emotion Recognition in Conversations","date":"2024-06-27","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/emotion-anchored-contrastive-learning","title":"Emotion-Anchored Contrastive Learning Framework for Emotion Recognition in Conversation","date":"2024-03-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/emodarts-joint-optimisation-of-cnn-sequential","title":"emoDARTS: Joint Optimisation of CNN & Sequential Neural Network Architectures for Superior Speech Emotion Recognition","date":"2024-03-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/ckerc-joint-large-language-models-with","title":"CKERC : Joint Large Language Models with Commonsense Knowledge for Emotion Recognition in Conversation","date":"2024-03-12","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/telme-teacher-leading-multimodal-fusion","title":"TelME: Teacher-leading Multimodal Fusion Network for Emotion Recognition in Conversation","date":"2024-01-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/joyful-joint-modality-fusion-and-graph","title":"Joyful: Joint Modality Fusion and Graph Contrastive Learning for Multimodal Emotion Recognition","date":"2023-11-18","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/accumulating-word-representations-in-multi","title":"Accumulating Word Representations in Multi-level Context Integration for ERC Task","date":"2023-11-06","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/a-transformer-based-model-with-self","title":"A Transformer-Based Model With Self-Distillation for Multimodal Emotion Recognition in Conversations","date":"2023-10-31","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/multimodal-prompt-transformer-with-hybrid","title":"Multimodal Prompt Transformer with Hybrid Contrastive Learning for Emotion Recognition in Conversation","date":"2023-10-04","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/instructerc-reforming-emotion-recognition-in","title":"InstructERC: Reforming Emotion Recognition in Conversation with Multi-task Retrieval-Augmented Large Language Models","date":"2023-09-21","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/revisiting-disentanglement-and-fusion-on","title":"Revisiting Disentanglement and Fusion on Modality and Context in Conversational Multimodal Emotion Recognition","date":"2023-08-08","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/cfn-esa-a-cross-modal-fusion-network-with","title":"CFN-ESA: A Cross-Modal Fusion Network with Emotion-Shift Awareness for Dialogue Emotion Recognition","date":"2023-07-28","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/fatrer-full-attention-topic-regularizer-for","title":"FATRER: Full-Attention Topic Regularizer for Accurate and Robust Conversational Emotion Recognition","date":"2023-07-23","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/supervised-adversarial-contrastive-learning","title":"Supervised Adversarial Contrastive Learning for Emotion Recognition in Conversations","date":"2023-06-02","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/improving-speech-emotion-recognition","title":"Enhancing Speech Emotion Recognition Through Differentiable Architecture Search","date":"2023-05-23","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/speech-text-dialog-pre-training-for-spoken","title":"Speech-Text Dialog Pre-training for Spoken Dialog Understanding with Explicit Cross-Modal Alignment","date":"2023-05-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/emotionic-emotional-inertia-and-contagion","title":"EmotionIC: emotional inertia and contagion-driven dependency modeling for emotion recognition in conversation","date":"2023-03-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/unimse-towards-unified-multimodal-sentiment","title":"UniMSE: Towards Unified Multimodal Sentiment Analysis and Emotion Recognition","date":"2022-11-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/supervised-prototypical-contrastive-learning","title":"Supervised Prototypical Contrastive Learning for Emotion Recognition in Conversation","date":"2022-10-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/ga2mif-graph-and-attention-based-two-stage","title":"GA2MIF: Graph and Attention Based Two-Stage Multi-Source Information Fusion for Conversational Emotion Detection","date":"2022-07-25","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/graphcfc-a-directed-graph-based-cross-modal","title":"GraphCFC: A Directed Graph Based Cross-Modal Feature Complementation Approach for Multimodal Conversational Emotion Recognition","date":"2022-07-06","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/m2fnet-multi-modal-fusion-network-for-emotion","title":"M2FNet: Multi-modal Fusion Network for Emotion Recognition in Conversation","date":"2022-06-05","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/emocaps-emotion-capsule-based-model-for","title":"EmoCaps: Emotion Capsule based Model for Conversational Emotion Recognition","date":"2022-03-25","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/mm-dfn-multimodal-dynamic-fusion-network-for","title":"MM-DFN: Multimodal Dynamic Fusion Network for Emotion Recognition in Conversations","date":"2022-03-04","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":1,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/structure-aware-transformer-for-graph","title":"Structure-Aware Transformer for Graph Representation Learning","date":"2022-02-07","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":4,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/speaker-normalization-for-self-supervised","title":"Speaker Normalization for Self-supervised Speech Emotion Recognition","date":"2022-02-02","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/s-page-a-speaker-and-position-aware-graph","title":"S+PAGE: A Speaker and Position-Aware Graph Neural Network Model for Emotion Recognition in Conversation","date":"2021-12-23","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/hybrid-curriculum-learning-for-emotion","title":"Hybrid Curriculum Learning for Emotion Recognition in Conversation","date":"2021-12-22","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/a-fine-tuned-wav2vec-2-0-hubert-benchmark-for","title":"A Fine-tuned Wav2vec 2.0/HuBERT Benchmark For Speech Emotion Recognition, Speaker Verification and Spoken Language Understanding","date":"2021-11-04","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/past-present-and-future-conversational","title":"Past, Present, and Future: Conversational Emotion Recognition through Structural Modeling of Psychological Knowledge","date":"2021-11-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/knowledge-interactive-network-with-sentiment","title":"Knowledge-Interactive Network with Sentiment Polarity Intensity-Aware Multi-Task Learning for Emotion Recognition in Conversations","date":"2021-11-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/a-discourse-aware-graph-neural-network-for","title":"A Discourse-Aware Graph Neural Network for Emotion Recognition in Multi-Party Conversation","date":"2021-11-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/speech-emotion-recognition-with-multi-task","title":"Speech Emotion Recognition with Multi-Task Learning","date":"2021-09-06","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/emoberta-speaker-aware-emotion-recognition-in","title":"EmoBERTa: Speaker-Aware Emotion Recognition in Conversation with RoBERTa","date":"2021-08-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/compm-context-modeling-with-speaker-s-pre","title":"CoMPM: Context Modeling with Speaker's Pre-trained Memory Tracking for Emotion Recognition in Conversation","date":"2021-08-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/dialoguecrn-contextual-reasoning-networks-for","title":"DialogueCRN: Contextual Reasoning Networks for Emotion Recognition in Conversations","date":"2021-06-03","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/topic-driven-and-knowledge-aware-transformer","title":"Topic-Driven and Knowledge-Aware Transformer for Dialogue Emotion Detection","date":"2021-06-02","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/coin-conversational-interactive-networks-for","title":"COIN: Conversational Interactive Networks for Emotion Recognition in Conversation","date":"2021-06-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/directed-acyclic-graph-network-for","title":"Directed Acyclic Graph Network for Conversational Emotion Recognition","date":"2021-05-27","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/a-hierarchical-transformer-with-speaker","title":"A Hierarchical Transformer with Speaker Modeling for Emotion Recognition in Conversation","date":"2020-12-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/dialogxl-all-in-one-xlnet-for-multi-party","title":"DialogXL: All-in-One XLNet for Multi-Party Conversation Emotion Recognition","date":"2020-12-16","rows_on_this_dataset":1,"code_links":4,"syntology":null},{"paper":"/paper/summarize-before-aggregate-a-global-to-local","title":"Summarize before Aggregate: A Global-to-local Heterogeneous Graph Inference Network for Conversational Emotion Recognition","date":"2020-12-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/hitrans-a-transformer-based-context-and","title":"HiTrans: A Transformer-Based Context- and Speaker-Sensitive Model for Emotion Detection in Conversations","date":"2020-12-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/an-iterative-emotion-interaction-network-for","title":"An Iterative Emotion Interaction Network for Emotion Recognition in Conversations","date":"2020-12-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/relation-aware-graph-attention-networks-with","title":"Relation-aware Graph Attention Networks with Relational Position Encodings for Emotion Recognition in Conversations","date":"2020-11-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/empirical-interpretation-of-speech-emotion","title":"Empirical Interpretation of Speech Emotion Perception with Attention Based Model for Speech Emotion Recognition","date":"2020-10-28","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/cosmic-commonsense-knowledge-for-emotion","title":"COSMIC: COmmonSense knowledge for eMotion Identification in Conversations","date":"2020-10-06","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/hierarchical-pre-training-for-sequence","title":"Hierarchical Pre-training for Sequence Labelling in Spoken Dialog","date":"2020-09-23","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/contextualized-emotion-recognition-in","title":"Contextualized Emotion Recognition in Conversation as Sequence Tagging","date":"2020-07-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/bieru-bidirectional-emotional-recurrent-unit","title":"BiERU: Bidirectional Emotional Recurrent Unit for Conversational Sentiment Analysis","date":"2020-05-31","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/real-time-emotion-recognition-via-attention","title":"Real-Time Emotion Recognition via Attention Gated Hierarchical Memory Network","date":"2019-11-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/speech-emotion-recognition-using-speech","title":"Speech Emotion Recognition Using Speech Feature and Word Embedding","date":"2019-11-18","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/emotion-recognition-in-conversations-with","title":"Conversational Transfer Learning for Emotion Recognition","date":"2019-10-11","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/knowledge-enriched-transformer-for-emotion","title":"Knowledge-Enriched Transformer for Emotion Detection in Textual Conversations","date":"2019-09-24","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dialoguegcn-a-graph-convolutional-neural","title":"DialogueGCN: A Graph Convolutional Neural Network for Emotion Recognition in Conversation","date":"2019-08-30","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/integrating-recurrence-dynamics-for-speech","title":"Integrating Recurrence Dynamics for Speech Emotion Recognition","date":"2018-11-09","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/dialoguernn-an-attentive-rnn-for-emotion","title":"DialogueRNN: An Attentive RNN for Emotion Detection in Conversations","date":"2018-11-01","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/icon-interactive-conversational-memory","title":"ICON: Interactive Conversational Memory Network for Multimodal Emotion Detection","date":"2018-10-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/conversational-memory-network-for-emotion","title":"Conversational Memory Network for Emotion Recognition in Dyadic Dialogue Videos","date":"2018-06-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/cnnlstm-architecture-for-speech-emotion","title":"CNN+LSTM Architecture for Speech Emotion Recognition with Data Augmentation","date":"2018-02-15","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/context-dependent-sentiment-analysis-in-user","title":"Context-Dependent Sentiment Analysis in User-Generated Videos","date":"2017-07-01","rows_on_this_dataset":1,"code_links":2,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":7,"samples_harvested":29,"samples_ran":14,"samples_unverified":15,"pointer_only_for_licence":6,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}