@article{Zheng2025, 
author = {Ruiqing Zheng and Yongxin He and Jiawen Huang and Shichao Kan and Hui Wang and Edwin Wang and Min Li},
title = {A Flexible Data-Driven Framework for Correcting Coarsely Annotated scRNA-seq Data},
year = {2025},
journal = {Big Data Mining and Analytics},
volume = {8},
number = {5},
pages = {997-1010},
keywords = {single-cell RNA sequencing (scRNA-seq), cell heterogeneity, cell annotation, supervised contrastive learning},
url = {https://www.sciopen.com/article/10.26599/BDMA.2025.9020009},
doi = {10.26599/BDMA.2025.9020009},
abstract = {Cells are the fundamental units of life and exhibit significant diversity in structure, behavior, and function, known as cell heterogeneity. The advent and development of single-cell RNA sequencing (scRNA-seq) technology have provided a crucial data foundation for studying cellular heterogeneity. Currently, most computational methods based on scRNA-seq involve a sequential process of clustering followed by annotation. However, those clustering-based methods are susceptible to the selection of genes and clustering parameters, resulting in inaccuracies in cell annotation. To address this issue, we develop a flexible data-driven cell correction framework based on partially annotated scRNA-seq data. This framework employs a neighborhood purity strategy and global selection strategies to select the anchor cells. Then, it optimizes a prediction neural network model using a classification loss with a contrastive regularization term to correct the labels of the remaining cells. The validity of this correction framework is demonstrated through various assessments on real scRNA-seq datasets. Based on the correct labels of scRNA-seq data, we further assess the latest unsupervised clustering methods, thereby establishing a more objective benchmark to compare their performance.}
}