@article{Zhang2026, 
author = {Wenxuan Zhang and Jihong Ding and Huazhong Liu and Tingting Han and Yuanhan Liu and Laurence T. Yang},
title = {Improving Cross-Modal Semantic Alignment with Cross-Modal Joint Semantic Transformer for Multimodal Sentiment Analysis},
year = {2026},
journal = {Big Data Mining and Analytics},
volume = {9},
number = {5},
pages = {1341-1353},
keywords = {Multimodal Sentiment Analysis (MSA), multimodal transformer, cross-modal semantic alignment, Singular Value Decomposition (SVD), Low-rank Multimodal Fusion (LMF)},
url = {https://www.sciopen.com/article/10.26599/BDMA.2025.9020110},
doi = {10.26599/BDMA.2025.9020110},
abstract = {Multimodal Sentiment Analysis (MSA) aims to comprehensively understand human affective states. To achieve this goal, it integrates heterogeneous modalities, including text, audio, and visual information. However, semantic misalignment within multimodal data and the insufficiency of multimodal feature fusion pose challenges to achieving accurate sentiment prediction. To this end, we propose a Cross-modal Joint Semantic Transformer (CJST) model to achieve cross-modal semantic alignment, thereby enhancing sentiment prediction accuracy. First, we design a Singular Value Decomposition (SVD) based cross-modal semantic alignment strategy that can decouple the time and semantic components of unimodal inputs to reduce the impact of misalignment noise and temporal redundancy. Then, a feature-level low-rank multimodal fusion strategy is developed to achieve high-order interactions among semantic features through tensor-based fusion within the low-rank space. Finally, we conduct various experiments on two well-known MSA benchmark datasets. Extensive experimental results indicate that the proposed CJST model outperforms or matches the state-of-the-art methods.}
}