{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/s-page-a-speaker-and-position-aware-graph","title":"S+PAGE: A Speaker and Position-Aware Graph Neural Network Model for Emotion Recognition in Conversation","arxiv_id":"2112.12389","date":"2021-12-23","proceeding":null,"authors":["Chen Liang","Chong Yang","Jing Xu","Juyang Huang","Yongliang Wang","Yang Dong"],"abstract":"Emotion recognition in conversation (ERC) has attracted much attention in recent years for its necessity in widespread applications. Existing ERC methods mostly model the self and inter-speaker context separately, posing a major issue for lacking enough interaction between them. In this paper, we propose a novel Speaker and Position-Aware Graph neural network model for ERC (S+PAGE), which contains three stages to combine the benefits of both Transformer and relational graph convolution network (R-GCN) for better contextual modeling. Firstly, a two-stream conversational Transformer is presented to extract the coarse self and inter-speaker contextual features for each utterance. Then, a speaker and position-aware conversation graph is constructed, and we propose an enhanced R-GCN model, called PAG, to refine the coarse features guided by a relative positional encoding. Finally, both of the features from the former two stages are input into a conditional random field layer to model the emotion transfer.","url_abs":"https://arxiv.org/abs/2112.12389v1","url_pdf":"https://arxiv.org/pdf/2112.12389v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[],"tasks":[{"task_slug":"emotion-recognition","task_name":"Emotion Recognition"},{"task_slug":"emotion-recognition-in-conversation","task_name":"Emotion Recognition in Conversation"},{"task_slug":"graph-neural-network","task_name":"Graph Neural Network"},{"task_slug":null,"task_name":"Position"}],"methods":[{"method_slug":"absolute-position-encodings","method_name":"Absolute Position Encodings"},{"method_slug":"adam","method_name":"Adam"},{"method_slug":"attention","method_name":"Attention"},{"method_slug":"bpe","method_name":"BPE"},{"method_slug":"convolution","method_name":"Convolution"},{"method_slug":"dense-connections","method_name":"Dense Connections"},{"method_slug":"dropout","method_name":"Dropout"},{"method_slug":"graph-neural-network","method_name":"Graph Neural Network"},{"method_slug":"label-smoothing","method_name":"Label Smoothing"},{"method_slug":"layer-normalization","method_name":"Layer Normalization"},{"method_slug":"linear-layer","method_name":"Linear Layer"},{"method_slug":"multi-head-attention","method_name":"Multi-Head Attention"},{"method_slug":"position-wise-feed-forward-layer","method_name":"Position-Wise Feed-Forward Layer"},{"method_slug":"residual-connection","method_name":"Residual Connection"},{"method_slug":"softmax","method_name":"Softmax"},{"method_slug":"transformer","method_name":"Transformer"}],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/emotion-recognition-in-conversation-on-3","task":"Emotion Recognition in Conversation","dataset":"DailyDialog","model":"S+PAGE","rank_in_archive_order":1,"of":22,"metrics":{"Micro-F1":"64.07"},"uses_additional_data":false},{"leaderboard":"/sota/emotion-recognition-in-conversation-on-4","task":"Emotion Recognition in Conversation","dataset":"EmoryNLP","model":"S+PAGE","rank_in_archive_order":12,"of":28,"metrics":{"Weighted-F1":"39.14"},"uses_additional_data":false},{"leaderboard":"/sota/emotion-recognition-in-conversation-on","task":"Emotion Recognition in Conversation","dataset":"IEMOCAP","model":"S+PAGE","rank_in_archive_order":26,"of":59,"metrics":{"Weighted-F1":"68.72"},"uses_additional_data":false},{"leaderboard":"/sota/emotion-recognition-in-conversation-on-meld","task":"Emotion Recognition in Conversation","dataset":"MELD","model":"S+PAGE","rank_in_archive_order":44,"of":68,"metrics":{"Weighted-F1":"63.32"},"uses_additional_data":false}],"syntology":{"syntology_url":"https://syntology.ai/paper/2112.12389","atlas_url":"https://app.syntology.ai/?focus=2112.12389","mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}