@article{Nayl2026, 
author = {Heba Nayl and Elkhateeb S. Aly and Amira Rezk and M. E. Fares},
title = {Improved chi-square feature selection for robust heart disease data classification},
year = {2026},
journal = {AIMS Mathematics},
volume = {11},
number = {1},
pages = {2682-2701},
keywords = {heart disease prediction, feature selection, chi-square statistical test, naive Bayes, classification performance},
url = {https://www.sciopen.com/article/10.3934/math.2026108},
doi = {10.3934/math.2026108},
abstract = {Early diagnosis of heart disease is vital for reducing mortality and improving patient outcomes; yet, accurate prediction remains a significant challenge owing to the complexity and high dimensionality of medical data. Data preprocessing is essential for overcoming these issues by cleaning, transforming, reducing, and balancing data to provide reliable inputs for feature selection and classification. This study introduces an improved chi-square (       χ          2      ) feature selection framework combined with multiple classifiers to enhance predictive performance. Our method was applied to Cleveland heart disease and diabetes datasets, where numeric attributes were discretized into categorical values, enabling        χ          2       to select the most informative features while eliminating redundancy. Several classifiers, including support vector machine (SVM), logistic regression (LR), K-nearest neighbors (KNN), and naive Bayes (NB), were trained using both the reduced subset and the complete feature set. Results show that the preprocessing include        χ          2       feature selection, achieved the highest performance. On the Cleveland dataset, the model attained a mean accuracy of 93.72%, precision of 94.01%, recall of 93.72%, F1-score of 93.74%, and an area under the curve(AUC) of 97.87%, while on the diabetes dataset, it achieved mean values of 93.55% accuracy, 94.23% precision, 93.55% recall, 93.48% F1-score, and an AUC 93.53%. The main contribution of this work lies in integrating discretization with        χ          2       based selection to produce a compact and discriminative feature subset. With a minimal number of selected features, the proposed approach delivers robust, accurate, and computationally efficient heart disease prediction, outperforming existing methods.}
}