@article{Ahmadi2026, 
author = {Ehsan Ahmadi and Asif Faisal Chowdhury and Chao Wang and Mahmood Jasim},
title = {Exoskeleton locomotion mode prediction in construction using GPT-4o: Zero-shot learning from vision and speech},
year = {2026},
journal = {Journal of Intelligent Construction},
keywords = {locomotion prediction, exoskeleton, multimodal sensor fusion, computer vision, natural language processing},
url = {https://www.sciopen.com/article/10.26599/JIC.2026.9180128},
doi = {10.26599/JIC.2026.9180128},
abstract = {Wearable exoskeletons enhance mobility and support in demanding tasks but face challenges in adapting to dynamic construction environments, particularly in predicting locomotion modes for tasks such as ladder climbing, stair navigation, low-space movement, and obstacle navigation. This study investigates the effectiveness of integrating speech and vision data for locomotion prediction while evaluating the generalization capability of large language models, specifically GPT-4o, through zero-shot learning compared to supervised fine-tuning. Using a multimodal framework with field-of-view frames and speech commands captured by smart glasses, we tested contrastive language-image pre-training (CLIP), ImageBind, and GPT-4o. Fine-tuned CLIP achieved an F1-score of 90.05% , yet GPT-4o’s zero-shot of 87.87% closely rivaled it, demonstrating strong adaptability to construction’s complex demands without task-specific training, while fine-tuned ImageBind trailed at 78.84%. This comparison underscores GPT-4o’s substantial potential to enable scalable exoskeleton control by leveraging multimodal comprehension of vision and speech data in dynamic construction settings.}
}