{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-emotion-recognition/papers/2","list_of":"/task/speech-emotion-recognition","task":"Speech Emotion Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":5,"rows_per_page":100,"rows":[101,200],"of":431,"counts":{"archive_papers_tagged":431,"with_a_code_link":139,"where_syntology_ran_a_sample":15,"not_listed_spam_title":0,"listed":431,"listed_where_code_ran":15,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":14,"every_run_a_failure_of_syntologys_instrument":1,"listed_with_a_run_with_no_instrument_failure":14,"listed_every_run_a_failure_of_syntologys_instrument":1,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-emotion-recognition","prev":"/task/speech-emotion-recognition","next":"/task/speech-emotion-recognition/papers/3","papers":[{"url":"/paper/dawn-of-the-transformer-era-in-speech-emotion","slug":"dawn-of-the-transformer-era-in-speech-emotion","title":"Dawn of the transformer era in speech emotion recognition: closing the valence gap","date":"2022-03-14","arxiv_id":"2203.07378","repositories_listed":1,"syntology":null},{"url":"/paper/privacy-preserving-speech-emotion-recognition","slug":"privacy-preserving-speech-emotion-recognition","title":"Privacy-preserving Speech Emotion Recognition through Semi-Supervised Federated Learning","date":"2022-02-05","arxiv_id":"2202.02611","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-graphs-for-audio","slug":"self-supervised-graphs-for-audio","title":"Self-supervised Graphs for Audio Representation Learning with Limited Labeled Data","date":"2022-01-31","arxiv_id":"2202.00097","repositories_listed":1,"syntology":null},{"url":"/paper/a-proposal-for-multimodal-emotion-recognition","slug":"a-proposal-for-multimodal-emotion-recognition","title":"A proposal for Multimodal Emotion Recognition using aural transformers and Action Units on RAVDESS dataset","date":"2021-12-30","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/attribute-inference-attack-of-speech-emotion","slug":"attribute-inference-attack-of-speech-emotion","title":"Attribute Inference Attack of Speech Emotion Recognition in Federated Learning Settings","date":"2021-12-26","arxiv_id":"2112.13416","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-wav2vec-2-0-fine-tuning-for","slug":"exploring-wav2vec-2-0-fine-tuning-for","title":"Exploring Wav2vec 2.0 fine-tuning for improved speech emotion recognition","date":"2021-10-12","arxiv_id":"2110.06309","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/exploring-wav2vec-2-0-fine-tuning-for#ran","syntology_url":"https://syntology.ai/paper/2110.06309","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.06309"}},"official":{"repos":["b04901014/FT-w2v2-ser"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/arabic-speech-emotion-recognition-employing","slug":"arabic-speech-emotion-recognition-employing","title":"Arabic Speech Emotion Recognition Employing Wav2vec2.0 and HuBERT Based on BAVED Dataset","date":"2021-10-09","arxiv_id":"2110.04425","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-label-uncertainty-modeling-for","slug":"end-to-end-label-uncertainty-modeling-for","title":"End-To-End Label Uncertainty Modeling for Speech-based Arousal Recognition Using Bayesian Neural Networks","date":"2021-10-07","arxiv_id":"2110.03299","repositories_listed":1,"syntology":null},{"url":"/paper/light-sernet-a-lightweight-fully","slug":"light-sernet-a-lightweight-fully","title":"Light-SERNet: A lightweight fully convolutional neural network for speech emotion recognition","date":"2021-10-07","arxiv_id":"2110.03435","repositories_listed":1,"syntology":null},{"url":"/paper/speech-emotion-recognition-with-multi-task","slug":"speech-emotion-recognition-with-multi-task","title":"Speech Emotion Recognition with Multi-Task Learning","date":"2021-09-06","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-cross-lingual-speech-emotion-1","slug":"unsupervised-cross-lingual-speech-emotion-1","title":"Unsupervised Cross-Lingual Speech Emotion Recognition Using Pseudo Multilabel","date":"2021-08-19","arxiv_id":"2108.08663","repositories_listed":1,"syntology":null},{"url":"/paper/an-improved-stargan-for-emotional-voice","slug":"an-improved-stargan-for-emotional-voice","title":"An Improved StarGAN for Emotional Voice Conversion: Enhancing Voice Quality and Data Augmentation","date":"2021-07-18","arxiv_id":"2107.08361","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-speech-emotion-recognition-using","slug":"efficient-speech-emotion-recognition-using","title":"Efficient Speech Emotion Recognition Using Multi-Scale CNN and Attention","date":"2021-06-08","arxiv_id":"2106.04133","repositories_listed":1,"syntology":null},{"url":"/paper/emonet-a-transfer-learning-framework-for","slug":"emonet-a-transfer-learning-framework-for","title":"EmoNet: A Transfer Learning Framework for Multi-Corpus Speech Emotion Recognition","date":"2021-03-10","arxiv_id":"2103.08310","repositories_listed":1,"syntology":null},{"url":"/paper/pre-trained-deep-convolution-neural-network","slug":"pre-trained-deep-convolution-neural-network","title":"Pre-trained Deep Convolution Neural Network Model With Attention for Speech Emotion Recognition","date":"2021-03-02","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/lssed-a-large-scale-dataset-and-benchmark-for","slug":"lssed-a-large-scale-dataset-and-benchmark-for","title":"LSSED: a large-scale dataset and benchmark for speech emotion recognition","date":"2021-01-30","arxiv_id":"2102.01754","repositories_listed":1,"syntology":null},{"url":"/paper/fixed-maml-for-few-shot-classification-in","slug":"fixed-maml-for-few-shot-classification-in","title":"Fixed-MAML for Few Shot Classification in Multilingual Speech Emotion Recognition","date":"2021-01-05","arxiv_id":"2101.01356","repositories_listed":1,"syntology":null},{"url":"/paper/a-novel-policy-for-pre-trained-deep","slug":"a-novel-policy-for-pre-trained-deep","title":"A novel policy for pre-trained Deep Reinforcement Learning for Speech Emotion Recognition","date":"2021-01-04","arxiv_id":"2101.00738","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-arabic-emotion-recognition-using","slug":"efficient-arabic-emotion-recognition-using","title":"Efficient Arabic emotion recognition using deep neural networks","date":"2020-10-31","arxiv_id":"2011.00346","repositories_listed":1,"syntology":null},{"url":"/paper/speech-simclr-combining-contrastive-and","slug":"speech-simclr-combining-contrastive-and","title":"Speech SIMCLR: Combining Contrastive and Reconstruction Objective for Self-supervised Speech Representation Learning","date":"2020-10-27","arxiv_id":"2010.13991","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/speech-simclr-combining-contrastive-and#ran","syntology_url":"https://syntology.ai/paper/2010.13991","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.13991"}},"official":{"repos":["athena-team/athena"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/is-everything-fine-grandma-acoustic-and","slug":"is-everything-fine-grandma-acoustic-and","title":"Is Everything Fine, Grandma? Acoustic and Linguistic Modeling for Robust Elderly Speech Emotion Recognition","date":"2020-09-07","arxiv_id":"2009.03432","repositories_listed":1,"syntology":null},{"url":"/paper/jointly-fine-tuning-bert-like-self-supervised-1","slug":"jointly-fine-tuning-bert-like-self-supervised-1","title":"Jointly Fine-Tuning “BERT-like” Self Supervised Models to Improve Multimodal Speech Emotion Recognition","date":"2020-08-15","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/deep-multilayer-perceptrons-for-dimensional","slug":"deep-multilayer-perceptrons-for-dimensional","title":"Deep Multilayer Perceptrons for Dimensional Speech Emotion Recognition","date":"2020-04-06","arxiv_id":"2004.02355","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-differences-between-song-and-speech","slug":"on-the-differences-between-song-and-speech","title":"On The Differences Between Song and Speech Emotion Recognition: Effect of Feature Sets, Feature Types, and Classifiers","date":"2020-04-01","arxiv_id":"2004.00200","repositories_listed":1,"syntology":null},{"url":"/paper/evaluation-of-error-and-correlation-based","slug":"evaluation-of-error-and-correlation-based","title":"Evaluation of Error and Correlation-Based Loss Functions For Multitask Learning Dimensional Speech Emotion Recognition","date":"2020-03-24","arxiv_id":"2003.10724","repositories_listed":1,"syntology":null},{"url":"/paper/speech-emotion-recognition-with-deep","slug":"speech-emotion-recognition-with-deep","title":"Speech emotion recognition with deep convolutional neural networks","date":"2020-02-15","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/non-linear-neurons-with-human-like-apical","slug":"non-linear-neurons-with-human-like-apical","title":"Non-linear Neurons with Human-like Apical Dendrite Activations","date":"2020-02-02","arxiv_id":"2003.03229","repositories_listed":1,"syntology":null},{"url":"/paper/attentive-modality-hopping-mechanism-for","slug":"attentive-modality-hopping-mechanism-for","title":"Attentive Modality Hopping Mechanism for Speech Emotion Recognition","date":"2019-11-29","arxiv_id":"1912.00846","repositories_listed":1,"syntology":null},{"url":"/paper/speech-emotion-recognition-using-speech","slug":"speech-emotion-recognition-using-speech","title":"Speech Emotion Recognition Using Speech Feature and Word Embedding","date":"2019-11-18","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-alignment-for-multimodal-emotion","slug":"learning-alignment-for-multimodal-emotion","title":"Learning Alignment for Multimodal Emotion Recognition from Speech","date":"2019-09-06","arxiv_id":"1909.05645","repositories_listed":1,"syntology":null},{"url":"/paper/an-interaction-aware-attention-network-for","slug":"an-interaction-aware-attention-network-for","title":"An Interaction-aware Attention Network for Speech Emotion Recognition in Spoken Dialogs","date":"2019-04-17","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/attention-augmented-end-to-end-multi-task","slug":"attention-augmented-end-to-end-multi-task","title":"Attention-Augmented End-to-End Multi-Task Learning for Emotion Prediction from Speech","date":"2019-03-29","arxiv_id":"1903.12424","repositories_listed":1,"syntology":null},{"url":"/paper/visualization-and-interpretation-of-latent","slug":"visualization-and-interpretation-of-latent","title":"Visualization and Interpretation of Latent Spaces for Controlling Expressive Speech Synthesis through Audio Analysis","date":"2019-03-27","arxiv_id":"1903.11570","repositories_listed":1,"syntology":null},{"url":"/paper/cross-lingual-speech-emotion-recognition-urdu","slug":"cross-lingual-speech-emotion-recognition-urdu","title":"Cross Lingual Speech Emotion Recognition: Urdu vs. Western Languages","date":"2018-12-15","arxiv_id":"1812.10411","repositories_listed":1,"syntology":null},{"url":"/paper/integrating-recurrence-dynamics-for-speech","slug":"integrating-recurrence-dynamics-for-speech","title":"Integrating Recurrence Dynamics for Speech Emotion Recognition","date":"2018-11-09","arxiv_id":"1811.04133","repositories_listed":1,"syntology":null},{"url":"/paper/the-emotional-voices-database-towards","slug":"the-emotional-voices-database-towards","title":"The Emotional Voices Database: Towards Controlling the Emotion Dimension in Voice Generation Systems","date":"2018-06-25","arxiv_id":"1806.09514","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/the-emotional-voices-database-towards#ran","syntology_url":"https://syntology.ai/paper/1806.09514","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.09514"}},"official":{"repos":["numediart/EmoV-DB"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/evaluating-gammatone-frequency-cepstral","slug":"evaluating-gammatone-frequency-cepstral","title":"Evaluating Gammatone Frequency Cepstral Coefficients with Neural Networks for Emotion Recognition from Speech","date":"2018-06-23","arxiv_id":"1806.09010","repositories_listed":1,"syntology":null},{"url":"/paper/attention-based-fully-convolutional-network","slug":"attention-based-fully-convolutional-network","title":"Attention Based Fully Convolutional Network for Speech Emotion Recognition","date":"2018-06-05","arxiv_id":"1806.01506","repositories_listed":1,"syntology":null},{"url":"/paper/transfer-learning-for-improving-speech","slug":"transfer-learning-for-improving-speech","title":"Transfer Learning for Improving Speech Emotion Classification Accuracy","date":"2018-01-19","arxiv_id":"1801.06353","repositories_listed":1,"syntology":null},{"url":null,"slug":"mater-multi-level-acoustic-and-textual","title":"MATER: Multi-level Acoustic and Textual Emotion Representation for Interpretable Speech Emotion Recognition","date":"2025-06-24","arxiv_id":"2506.19887","repositories_listed":0,"syntology":null},{"url":null,"slug":"developing-a-high-performance-framework-for","title":"Developing a High-performance Framework for Speech Emotion Recognition in Naturalistic Conditions Challenge for Emotional Attribute Prediction","date":"2025-06-12","arxiv_id":"2506.10930","repositories_listed":0,"syntology":null},{"url":null,"slug":"co-vada-a-confidence-oriented-voice","title":"CO-VADA: A Confidence-Oriented Voice Augmentation Debiasing Approach for Fair Speech Emotion Recognition","date":"2025-06-06","arxiv_id":"2506.06071","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-04652","title":"EMO-Debias: Benchmarking Gender Debiasing Techniques in Multi-Label Speech Emotion Recognition","date":"2025-06-05","arxiv_id":"2506.04652","repositories_listed":0,"syntology":null},{"url":null,"slug":"hyfuse-aligning-heterogeneous-speech-pre","title":"HYFuse: Aligning Heterogeneous Speech Pre-Trained Representations in Hyperbolic Space for Speech Emotion Recognition","date":"2025-06-03","arxiv_id":"2506.03403","repositories_listed":0,"syntology":null},{"url":null,"slug":"are-mamba-based-audio-foundation-models-the","title":"Are Mamba-based Audio Foundation Models the Best Fit for Non-Verbal Emotion Recognition?","date":"2025-06-02","arxiv_id":"2506.02258","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-speech-emotion-recognition-with","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","date":"2025-06-02","arxiv_id":"2506.02088","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-the-impact-of-word","title":"Investigating the Impact of Word Informativeness on Speech Emotion Recognition","date":"2025-06-02","arxiv_id":"2506.02239","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-machine-unlearning-for-paralinguistic","title":"Towards Machine Unlearning for Paralinguistic Speech Processing","date":"2025-06-02","arxiv_id":"2506.02230","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-more-with-less-self-supervised","title":"Learning More with Less: Self-Supervised Approaches for Low-Resource Speech Emotion Recognition","date":"2025-06-01","arxiv_id":"2506.02059","repositories_listed":0,"syntology":null},{"url":null,"slug":"parrot-synergizing-mamba-and-attention-based","title":"PARROT: Synergizing Mamba and Attention-based SSL Pre-Trained Models via Parallel Branch Hadamard Optimal Transport for Speech Emotion Recognition","date":"2025-06-01","arxiv_id":"2506.01138","repositories_listed":0,"syntology":null},{"url":null,"slug":"source-tracing-of-synthetic-speech-systems","title":"Source Tracing of Synthetic Speech Systems Through Paralinguistic Pre-Trained Representations","date":"2025-06-01","arxiv_id":"2506.01157","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-emotion-fool-anti-spoofing","title":"Can Emotion Fool Anti-spoofing?","date":"2025-05-29","arxiv_id":"2505.23962","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-perser-few-shot-listener-personalized","title":"Meta-PerSER: Few-Shot Listener Personalized Speech Emotion Recognition via Meta-learning","date":"2025-05-22","arxiv_id":"2505.16220","repositories_listed":0,"syntology":null},{"url":null,"slug":"mitigating-subgroup-disparities-in-multi","title":"Mitigating Subgroup Disparities in Multi-Label Speech Emotion Recognition: A Pseudo-Labeling and Unsupervised Learning Approach","date":"2025-05-20","arxiv_id":"2505.14449","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-11051","title":"CAMEO: Collection of Multilingual Emotional Speech Corpora","date":"2025-05-16","arxiv_id":"2505.11051","repositories_listed":0,"syntology":null},{"url":null,"slug":"empirical-analysis-of-asynchronous-federated","title":"Empirical Analysis of Asynchronous Federated Learning on Heterogeneous Devices: Efficiency, Fairness, and Privacy Trade-offs","date":"2025-05-11","arxiv_id":"2505.07041","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-for-speech-emotion-recognition","title":"Deep Learning for Speech Emotion Recognition: A CNN Approach Utilizing Mel Spectrograms","date":"2025-03-25","arxiv_id":"2503.19677","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-calibrated-affective-speech-recognition","title":"Coverage-Guaranteed Speech Emotion Recognition via Calibrated Uncertainty-Adaptive Prediction Sets","date":"2025-03-24","arxiv_id":"2503.22712","repositories_listed":0,"syntology":null},{"url":null,"slug":"heterogeneous-bimodal-attention-fusion-for","title":"Heterogeneous bimodal attention fusion for speech emotion recognition","date":"2025-03-09","arxiv_id":"2503.06405","repositories_listed":0,"syntology":null},{"url":null,"slug":"bimodal-connection-attention-fusion-for","title":"Bimodal Connection Attention Fusion for Speech Emotion Recognition","date":"2025-03-08","arxiv_id":"2503.05858","repositories_listed":0,"syntology":null},{"url":null,"slug":"emoformer-a-text-independent-speech-emotion","title":"EmoFormer: A Text-Independent Speech Emotion Recognition using a Hybrid Transformer-CNN model","date":"2025-01-22","arxiv_id":"2501.12682","repositories_listed":0,"syntology":null},{"url":null,"slug":"emotech-a-multi-modal-speech-emotion","title":"EmoTech: A Multi-modal Speech Emotion Recognition Using Multi-source Low-level Information with Hybrid Recurrent Network","date":"2025-01-22","arxiv_id":"2501.12674","repositories_listed":0,"syntology":null},{"url":null,"slug":"parameterised-quantum-circuits-for-novel","title":"Representation Learning with Parameterised Quantum Circuits for Advancing Speech Emotion Recognition","date":"2025-01-21","arxiv_id":"2501.12050","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-cross-attention-transformer-and","title":"Leveraging Cross-Attention Transformer and Multi-Feature Fusion for Cross-Linguistic Speech Emotion Recognition","date":"2025-01-06","arxiv_id":"2501.10408","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-it-still-fair-investigating-gender","title":"Is It Still Fair? Investigating Gender Fairness in Cross-Corpus Speech Emotion Recognition","date":"2025-01-02","arxiv_id":"2501.00995","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-discriminative-features-from","title":"learning discriminative features from spectrograms using center loss for speech emotion recognition","date":"2025-01-02","arxiv_id":"2501.01103","repositories_listed":0,"syntology":null},{"url":null,"slug":"metadata-enhanced-speech-emotion-recognition","title":"Metadata-Enhanced Speech Emotion Recognition: Augmented Residual Integration and Co-Attention in Two-Stage Fine-Tuning","date":"2024-12-30","arxiv_id":"2412.20707","repositories_listed":0,"syntology":null},{"url":null,"slug":"mouth-articulation-based-anchoring-for","title":"Mouth Articulation-Based Anchoring for Improved Cross-Corpus Speech Emotion Recognition","date":"2024-12-27","arxiv_id":"2412.19909","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhanced-speech-emotion-recognition-with","title":"Enhanced Speech Emotion Recognition with Efficient Channel Attention Guided Deep CNN-BiLSTM Framework","date":"2024-12-13","arxiv_id":"2412.10011","repositories_listed":0,"syntology":null},{"url":null,"slug":"wavfusion-towards-wav2vec-2-0-multimodal","title":"WavFusion: Towards wav2vec 2.0 Multimodal Speech Emotion Recognition","date":"2024-12-07","arxiv_id":"2412.05558","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-cross-corpus-speech-emotion-recognition","title":"A Cross-Corpus Speech Emotion Recognition Method Based on Supervised Contrastive Learning","date":"2024-11-25","arxiv_id":"2411.19803","repositories_listed":0,"syntology":null},{"url":null,"slug":"once-more-with-feeling-measuring-emotion-of","title":"Once More, With Feeling: Measuring Emotion of Acting Performances in Contemporary American Film","date":"2024-11-15","arxiv_id":"2411.10018","repositories_listed":0,"syntology":null},{"url":null,"slug":"improvement-and-implementation-of-a-speech","title":"Improvement and Implementation of a Speech Emotion Recognition Model Based on Dual-Layer LSTM","date":"2024-11-14","arxiv_id":"2411.09189","repositories_listed":0,"syntology":null},{"url":null,"slug":"re-parameterization-of-lightweight","title":"Re-Parameterization of Lightweight Transformer for On-Device Speech Emotion Recognition","date":"2024-11-14","arxiv_id":"2411.09339","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-speech-based-emotion-recognition","title":"Improving Speech-based Emotion Recognition with Contextual Utterance Analysis and LLMs","date":"2024-10-27","arxiv_id":"2410.20334","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-speech-large-language-models","title":"A Survey on Speech Large Language Models","date":"2024-10-24","arxiv_id":"2410.18908","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-effective-speaker-property","title":"Investigating Effective Speaker Property Privacy Protection in Federated Learning for Speech Emotion Recognition","date":"2024-10-17","arxiv_id":"2410.13221","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-speech-emotion-recognition-through","title":"Enhancing Speech Emotion Recognition through Segmental Average Pooling of Self-Supervised Learning Features","date":"2024-10-16","arxiv_id":"2410.12416","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-view-multi-task-modeling-with-speech","title":"Multi-View Multi-Task Modeling with Speech Foundation Models for Speech Forensic Tasks","date":"2024-10-16","arxiv_id":"2410.12947","repositories_listed":0,"syntology":null},{"url":null,"slug":"sequifi-mitigating-catastrophic-forgetting-in","title":"SeQuiFi: Mitigating Catastrophic Forgetting in Speech Emotion Recognition with Sequential Class-Finetuning","date":"2024-10-16","arxiv_id":"2410.12567","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-we-estimate-purchase-intention-based-on","title":"Can We Estimate Purchase Intention Based on Zero-shot Speech Emotion Recognition?","date":"2024-10-12","arxiv_id":"2410.09636","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-cross-lingual-meta-learning-method-based-on","title":"A Cross-Lingual Meta-Learning Method Based on Domain Adaptation for Speech Emotion Recognition","date":"2024-10-06","arxiv_id":"2410.04633","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-scale-temporal-transformer-for-speech","title":"Multi-Scale Temporal Transformer For Speech Emotion Recognition","date":"2024-10-01","arxiv_id":"2410.00390","repositories_listed":0,"syntology":null},{"url":null,"slug":"two-stage-framework-for-robust-speech-emotion","title":"Two-stage Framework for Robust Speech Emotion Recognition Using Target Speaker Extraction in Human Speech Noise Conditions","date":"2024-09-29","arxiv_id":"2409.19585","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-acoustic-similarity-in-emotional","title":"Exploring Acoustic Similarity in Emotional Speech and Music via Self-Supervised Representations","date":"2024-09-26","arxiv_id":"2409.17899","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-speech-emotion-recognition-in","title":"Personalized Speech Emotion Recognition in Human-Robot Interaction using Vision Transformers","date":"2024-09-16","arxiv_id":"2409.10687","repositories_listed":0,"syntology":null},{"url":null,"slug":"stimulus-modality-matters-impact-of","title":"Stimulus Modality Matters: Impact of Perceptual Evaluations from Different Modalities on Speech Emotion Recognition System Performance","date":"2024-09-16","arxiv_id":"2409.10762","repositories_listed":0,"syntology":null},{"url":null,"slug":"turbo-your-multi-modal-classification-with","title":"Turbo your multi-modal classification with contrastive learning","date":"2024-09-14","arxiv_id":"2409.09282","repositories_listed":0,"syntology":null},{"url":null,"slug":"consensus-based-distributed-quantum-kernel","title":"Consensus-based Distributed Quantum Kernel Learning for Speech Recognition","date":"2024-09-09","arxiv_id":"2409.05770","repositories_listed":0,"syntology":null},{"url":null,"slug":"searching-for-effective-preprocessing-method","title":"Searching for Effective Preprocessing Method and CNN-based Architecture with Efficient Channel Attention on Speech Emotion Recognition","date":"2024-09-06","arxiv_id":"2409.04007","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-enhancement-for-computer-audition-an","title":"Audio Enhancement for Computer Audition -- An Iterative Training Paradigm Using Sample Importance","date":"2024-08-12","arxiv_id":"2408.06264","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-03150","title":"Conditioning LLMs with Emotion in Neural Machine Translation","date":"2024-08-06","arxiv_id":"2408.03150","repositories_listed":0,"syntology":null},{"url":null,"slug":"describe-where-you-are-improving-noise","title":"Describe Where You Are: Improving Noise-Robustness for Speech Emotion Recognition with Text Description of the Environment","date":"2024-07-25","arxiv_id":"2407.17716","repositories_listed":0,"syntology":null},{"url":null,"slug":"emo-codec-a-depth-look-at-emotion","title":"EMO-Codec: An In-Depth Look at Emotion Preservation capacity of Legacy and Neural Codec Models With Subjective and Objective Evaluations","date":"2024-07-22","arxiv_id":"2407.15458","repositories_listed":0,"syntology":null},{"url":null,"slug":"pcq-emotion-recognition-in-speech-via","title":"PCQ: Emotion Recognition in Speech via Progressive Channel Querying","date":"2024-07-17","arxiv_id":"2407.12380","repositories_listed":0,"syntology":null},{"url":null,"slug":"msp-podcast-ser-challenge-2024-l-antenne-du","title":"MSP-Podcast SER Challenge 2024: L'antenne du Ventoux Multimodal Self-Supervised Learning for Speech Emotion Recognition","date":"2024-07-08","arxiv_id":"2407.05746","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-layer-anchoring-strategy-for-enhancing","title":"A Layer-Anchoring Strategy for Enhancing Cross-Lingual Speech Emotion Recognition","date":"2024-07-06","arxiv_id":"2407.04966","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-emotion-recognition-under-resource","title":"Breaking Resource Barriers in Speech Emotion Recognition via Data Distillation","date":"2024-06-21","arxiv_id":"2406.15119","repositories_listed":0,"syntology":null},{"url":null,"slug":"double-multi-head-attention-multimodal-system","title":"Double Multi-Head Attention Multimodal System for Odyssey 2024 Speech Emotion Recognition Challenge","date":"2024-06-15","arxiv_id":"2406.10598","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-emotion-recognition-using-cnn-and-its","title":"Speech Emotion Recognition Using CNN and Its Use Case in Digital Healthcare","date":"2024-06-15","arxiv_id":"2406.10741","repositories_listed":0,"syntology":null}],"record_sha256":"e7819604175a4a6c47434b22c4e54e84a85bf006da2a517a164a0c1e88c0bd1f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}