Copyright © 2026 Authors retain the copyright of this article. This article is an open access article distributed under the Creative Commons Attribution License which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.
@article{207946,
author = {Busthana Yasmin M.T. and Haris U. and Najiya K. T. and Irfana Febin K.},
title = {Supervised Machine Learning and Clinical Decision-Support Frameworks for Chronic Disease Diagnosis: A Comparative Multi-Domain Analysis},
journal = {International Journal of Innovative Research in Technology},
year = {2026},
volume = {13},
number = {3},
pages = {3545-3554},
issn = {2349-6002},
url = {https://ijirt.org/article?manuscript=207946},
abstract = {The increasing availability of structured healthcare data has accelerated the adoption of machine learning (ML) techniques for disease prediction and clinical decision support. This study presents a systematic comparative evaluation of six supervised machine learning algorithms - Logistic Regression (LR), Decision Tree (DT), Random Forest (RF), Support Vector Machine (SVM), K-Nearest Neighbours (KNN) and Naïve Bayes (NB) - for predicting breast cancer, diabetes and heart disease using benchmark clinical datasets. To ensure replicability and address data heterogeneity, standard pre-processing pipelines, including MinMax feature scaling and Synthetic Minority Oversampling Technique (SMOTE), were applied alongside stratified 10-fold cross-validation. Evaluation metrics include accuracy, precision, recall (sensitivity), F1-score, specificity and area under the receiver operating characteristic curve (AUC-ROC). Experimental results indicate that SVM achieved the highest diagnostic performance for high-dimensional breast cancer features (accuracy: 98.25%, recall: 1.0000, F1-score: 0.9861, AUC-ROC: 0.9912). For diabetes classification, SVM yielded superior results (accuracy: 74.16%, F1-score: 0.7294, AUC-ROC: 0.7920), effectively mitigating challenges associated with class imbalance and feature interdependency. In cardiovascular disease prediction, KNN demonstrated optimal local boundary classification (accuracy: 91.80%, F1-score: 0.9206, AUC-ROC: 0.9410). Random Forest exhibited stable performance across domains, whereas Logistic Regression provided transparent decision rules suitable for direct clinical interpretation. The findings underscore the domain-dependent behaviour of classifiers and highlight the critical balance required between predictive performance and explainability in clinical decision-support systems.},
keywords = {machine learning, clinical decision support, disease prediction, diagnostic performance, biomedical classification, health informatics, optimization.},
month = {August},
}
Submit your research paper and those of your network (friends, colleagues, or peers) through your IPN account, and receive 800 INR for each paper that gets published.
Join NowNational Conference on Sustainable Engineering and Management - 2024 Last Date: 15th March 2024
Submit inquiry