@article{Lee2026, 
author = {Seung-su Lee and Young-Been Noh and HwaYoung Jeong and Kwang-il Hwang},
title = {A Spatial-Temporal Normalized Contrastive Embedding for Robust Motion Similarity Retrieval},
year = {2026},
journal = {Computers, Materials & Continua},
volume = {88},
number = {3},
pages = {5},
keywords = {Motion similarity retrieval, spatial-temporal normalization, contrastive representation learning, temporal convolutional networks (TCN), prototype-based embedding},
url = {https://www.sciopen.com/article/10.32604/cmc.2026.081251},
doi = {10.32604/cmc.2026.081251},
abstract = {Robust motion similarity retrieval from monocular 2D pose sequences is challenged by body-scale variation, viewpoint inconsistency, translation drift, and temporal misalignment. Existing contrastive skeleton learning methods primarily address action recognition and rarely integrate explicit geometric canonicalization for retrieval-oriented metric learning. This paper proposes a spatial-temporal normalized contrastive embedding framework that unifies structured nuisance suppression with scalable similarity representation learning. A four-stage normalization pipeline—torso-scale normalization, pelvis-centered alignment, posture-axis alignment, and phase-synchronized temporal resampling—removes geometric and temporal distortions prior to embedding. The normalized sequences are encoded using an acausal dilated temporal convolutional network trained with a hybrid contrastive objective combining NT-Xent and semi-hard triplet loss, enabling both global separation and fine-grained stylistic discrimination. A prototype-based representation further supports interpretable amateur-to-professional style mapping. Experiments on a golf swing benchmark achieve a Top-1 accuracy of 91.3%, outperforming BiLSTM and Dynamic Time Warping baselines. The framework establishes an invariant and interpretable paradigm for motion similarity retrieval applicable to broader human movement analysis tasks.}
}