@article{Alshaya2026, 
author = {Hend Alshaya},
title = {Causal Counterfactual Transformers for Explainable Video-Based Action Recognition Based on CauFormer-V Framework},
year = {2026},
journal = {Computers, Materials & Continua},
volume = {88},
number = {3},
pages = {19},
keywords = {Causal inference, counterfactual learning, video representation learning, temporal transformer, action recognition, causal attention mechanism},
url = {https://www.sciopen.com/article/10.32604/cmc.2026.080758},
doi = {10.32604/cmc.2026.080758},
abstract = {Video representation learning faces very challenging goals, including spurious temporal correlations, confounding visual features, and failure to learn real causal relationships between video events. Current transformer-based approaches learn statistical relationships rather than causal interactions, leading to weak generalization and high sensitivity to distribution changes. The current paper proposes a new Counterfactual Transformer Network, named CauFormer-V, that combines causal inference concepts with temporal representation learning for video. The framework was proposed and includes three main innovations, (1) a Causal Temporal Attention (CTA) mechanism, a mechanism that specifically models causal dependencies among video frames via do-calculus intervention, (2) a Counterfactual Video Generator (CVG) module, which is used to generate counterfactual video representations to enable causal learning, and (3) a Temporal Causal Graph (TCG) network, which is a structure that explicitly models causal dependencies across multiple temporal scales. Prolonged testing on four benchmark datasets, Something-Something V2, Kinetics-400, UCF101, and HMDB51, shows that CauFormer-V yields the highest possible results of 74.8% Top-1 accuracy on Something-Something V2, 86.7% on Kinetics-400, 97.9% on UCF101, and 78.4% on HMDB51, outperforming other leading methods by 2.1%–4.3%. Ablation studies confirm the effectiveness of each component, whereas visualization analysis indicates that CauFormer-V can extract semantically relevant causal temporal patterns. The presented framework offers a principled approach to learning powerful video representations with higher interpretability and stronger generalization.}
}