@article{Alsafari2026, 
author = {Safa Alsafari and Ayman Yafoz},
title = {Few-Shot Bearing Fault Diagnosis under Joint Fault-Severity and Load Shift: A Leak-Free Cross-Domain Benchmark},
year = {2026},
journal = {Computer Modeling in Engineering & Sciences},
volume = {148},
number = {1},
pages = {17},
keywords = {Bearing fault diagnosis, capsule network, cross-domain adaptation, few-shot learning, double domain shift},
url = {https://www.sciopen.com/article/10.32604/cmes.2026.084403},
doi = {10.32604/cmes.2026.084403},
abstract = {Bearing fault diagnosis in industrial deployment must contend with two simultaneous distributional shifts: fault severity increases as damage progresses, and motors operate at loads unseen during training. We define this compound setting as the double domain shift and present a rigorous few-shot benchmark on the Case Western Reserve University (CWRU) and Paderborn University (PU) bearing datasets. Six architectures spanning distinct learning paradigms—a multilayer perceptron (MLP), a capsule network (CapsNet), a residual capsule network (ResCaps), a prototypical network (ProtoNet), a modified residual convolutional network (MRCN), and Deep Correlation Alignment (Deep CORAL)—are evaluated under a strict three-way split (support/validation/held-out test) that prevents the data-leakage patterns prevalent in prior CWRU protocols. Models are adapted using  K∈{5,10,20} labelled target samples and assessed on three complementary metrics: accuracy, macro F1-score, and Cohen’s  κ. A lightweight 1-D convolutional neural network (CNN) with a capsule routing head (CapsNet) leads on 15 of 18 CWRU conditions and on all PU conditions at  K ≥ 10 (with MRCN leading at PU Target-A  K=5), achieving macro F1 of  0.860 and  κ=0.818 at  K=5 on the harder CWRU target—with 18,624 parameters (roughly one-third of the MLP baseline) and without any distribution-alignment objective. The multi-metric evaluation reveals findings invisible to accuracy alone: several baselines fall below moderate agreement ( κ &lt; 0.60) at low  K, and MRCN’s accuracy–F1 gap of  5.3 percentage points (pp) at  K=10 exposes class-selective failure that accuracy conceals. On PU, a macro F1 of  0.443 at  K=5 on the real inner-race target quantifies the artificial-to-real fatigue transfer gap, and all models produce  κ &lt; 0.06 on the outer-race target at  K=5, establishing a realistic lower bound for future work. Wilcoxon signed-rank tests confirm the CapsNet advantage is statistically significant against the weaker baselines in nearly all conditions. Capsule output norms provide interpretable, per-class confidence-like activation scores without requiring post-hoc attribution methods.}
}