
@Article{cmes.2026.086256,
AUTHOR = {Muhammad Sheraz, Adil Majeed, Shehzad Khalid, Yazeed Alkhrijah, Sulieman S. Alshuhri, Hasan Mujtaba},
TITLE = {Multimodal Emotion Recognition in Urdu through Late Fusion of Fine-Tuned Speech and Text Representations},
JOURNAL = {Computer Modeling in Engineering \& Sciences},
VOLUME = {148},
YEAR = {2026},
NUMBER = {2},
PAGES = {--},
URL = {http://www.techscience.com/CMES/v148n2/68592},
ISSN = {1526-1506},
ABSTRACT = {Emotion recognition plays a crucial role in enabling intelligent human–computer interaction, yet research in low-resource languages such as Urdu remains limited, particularly in multimodal settings. This study proposes a multimodal deep learning framework for Urdu emotion recognition by integrating speech and text modalities. The approach leverages transformer-based models, namely wav2vec 2.0 for audio representation and MuRIL for text representation, combined using a late fusion strategy for classification. Experiments were conducted on the UMED dataset, consisting of 8269 multimodal instances across five emotion classes. The proposed multimodal model achieved an accuracy of 0.701 and an F1-score of 0.6915, outperforming unimodal baselines, where the audio-only and text-only models achieved accuracies of 0.6681 and 0.5085, respectively. Furthermore, the proposed approach surpasses the existing UMEDNet benchmark, demonstrating the effectiveness of transformer-based feature extraction and multimodal late fusion for Urdu emotion recognition. The results highlight the complementary nature of speech and text modalities and demonstrate that independently learned modality-specific classifiers combined through decision-level fusion can improve emotion recognition performance in low-resource languages. However, the performance improvement over alternative fusion strategies was relatively modest, indicating that more advanced multimodal interaction mechanisms may further enhance recognition performance.},
DOI = {10.32604/cmes.2026.086256}
}



