@inproceedings{8d966eacb2b04125b31d6f6fc5be588b,
title = "MAMF-Net: Modality-Adaptive Masked Fusion Network for Speech Emotion Recognition",
abstract = "This paper introduces a novel multimodal emotion recognition model, the Modality-Adaptive Masked Fusion Network (MAMF-Net), designed to mitigate information loss and improve cross-modal alignment during the fusion of speech and text modalities. MAMF-Net employs an audio-guided text encoder to enhance the semantic representation of text by leveraging the temporal resolution and contextual information inherent in speech, thereby ensuring accurate alignment of modal features. Additionally, the model utilizes a modality transfer-based MAE masking strategy, which effectively captures complementary information between modalities by partially masking transferred information, thus improving fusion effectiveness and system stability. The experimental results show that MAMF-Net outperforms existing methods on datasets such as CMU-MOSI and CMU-MOSEI, highlighting its significant potential for multimodal emotion analysis.",
keywords = "Speech emotion recognition, cross modality, masked autoencoder, multimodal fusion",
author = "Hengrui Li and Tianyi Lu and Jianfeng Wang and Xiaopei Chen and Yongbing Zhang and Shaohui Liu",
note = "Publisher Copyright: {\textcopyright} 2025 IEEE.; 2025 IEEE International Conference on Multimedia and Expo, ICME 2025 ; Conference date: 30-06-2025 Through 04-07-2025",
year = "2025",
doi = "10.1109/ICME59968.2025.11209516",
language = "英语",
series = "Proceedings - IEEE International Conference on Multimedia and Expo",
publisher = "IEEE Computer Society",
booktitle = "2025 IEEE International Conference on Multimedia and Expo",
address = "美国",
}