@article{YANG2026, 
author = {Junmei YANG and Bangcheng ZHANG and Lu YANG and Delu ZENG},
title = {A Single-Channel Speech Separation Model Based on Time-Domain Comprehensive Attention Mechanism},
year = {2026},
journal = {Journal of South China University of Technology (Natural Science Edition)},
volume = {54},
number = {1},
pages = {70-82},
keywords = {deep learning, speech separation, Transformer module, Conformer structure, comprehensive attention},
url = {https://www.sciopen.com/article/10.12141/j.issn.1000-565X.250054},
doi = {10.12141/j.issn.1000-565X.250054},
abstract = {Single-channel speech separation aims to extract clean target speaker speech from a mixed audio signal recorded by a single microphone, with significant application value in scenarios such as smart homes, conference systems, and hearing aids. With the rapid development of deep learning technology, self-attention network-based approaches to single-channel speech separation have achieved remarkable progress. While self-attention networks excel at capturing contextual information in long sequence, they still exhibit limitations in capturing detailed features such as temporal/spectral continuity, spectral structure, and timbre in real-world speech scenarios. Moreover, existing separation architectures based on a single attention paradigm struggle to achieve effective multi-scale feature fusion. To address these challenges, this paper proposed a Temporal Comprehensive Attention Network (TCANet), which addresses the aforementioned issues through a synergistic design of local and global attention modules. Local modeling employs an S&amp; C-SENet-enhanced Conformer structure to capture short-term features such as spectral structure and timbre in detail, while global modeling incorporates a modified Transformer module with relative position embedding to explicitly learn long-term speech dependencies in speech. Furthermore, TCANet achieves cross-scale fusion of intra-block local features and inter-block global correlations through a dimension transformation mechanism. Experimental results on three benchmark datasets—LRS2-2Mix, Libri2Mix, and EchoSet—demonstrate that the proposed method outperforms existing end-to-end speech separation approaches in terms of scale-invariant signal-to-noise ratio improvement (SI-SNRi) and signal-to-distortion ratio improvement (SDRi).}
}