@article{Luo2026, 
author = {Hanyu Luo and Jiayu Yuan and Xinping Li and Chang Liu and Jiuxin Cao},
title = {GloLoc: Contrastive Learning Enhanced Global-Local Fusion for Multimodal Aspect-Based Sentiment Analysis},
year = {2026},
journal = {Big Data Mining and Analytics},
keywords = {aspect-based sentiment analysis, multimodal fusion, contrastive learning, data augmentation},
url = {https://www.sciopen.com/article/10.26599/BDMA.2026.9020009},
doi = {10.26599/BDMA.2026.9020009},
abstract = {Multimodal aspect-based sentiment analysis (MABSA) addresses the task of predicting aspect-level sentiment polarity by jointly modeling textual semantics and visual information, which is particularly valuable for large-scale social media applications. Prior approaches mainly rely on direct aspect–object correspondence, often neglecting contextual signals and aesthetic factors, thereby limiting robustness and generalizability. This study proposes GloLoc, a global–local multimodal fusion architecture designed for aspect-based sentiment analysis, comprising two essential components: a Global Feature Extractor that leverages BLIP and VILA models to capture semantic content and aesthetic cues from images, and a Local Alignment Module that exploits scene graphs to represent inter-object relations for precise cross-modal alignment between aspects and visual regions. To further improve robustness against linguistic variation, we incorporate both supervised and self-supervised contrastive learning, the latter enhanced by a sentiment-preserving augmentation strategy (SentiAug). Experimental evaluations on benchmark datasets confirm the performance advantages of our approach.}
}