Copyright © 2026 Authors retain the copyright of this article. This article is an open access article distributed under the Creative Commons Attribution License which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.
@article{208124,
author = {Shahina Begum and Rubina Begum},
title = {Multimodal Deep Learning for Cardiovascular and Diabetes Risk Prediction Using Retinal Fundus Images and Clinical Data},
journal = {International Journal of Innovative Research in Technology},
year = {2026},
volume = {13},
number = {4},
pages = {324-330},
issn = {2349-6002},
url = {https://ijirt.org/article?manuscript=208124},
abstract = {cardiovascular disease and diabetes are major health conditions for which early risk assessment can support timely clinical evaluation. Retinal fundus images provide non-invasive information about retinal microvascular characteristics, while structured clinical variables provide complementary physiological information. This work presents a multimodal deep learning framework that combines retinal fundus images with clinical data for simultaneous cardiovascular and diabetes risk prediction. Retinal images are resized to 224×224 pixels and processed using an ImageNet-pretrained EfficientNet-B3 backbone. A two-layer Transformer Encoder with eight attention heads is applied to the extracted feature sequence to model global contextual relationships. Clinical variables comprising blood pressure, body mass index, glucose, cholesterol, age, and diabetes status are processed through a fully connected network. The resulting image and clinical representations are combined using an attention-based fusion mechanism, followed by two task-specific output heads. Grad-CAM is incorporated to provide visual explanations of image-based predictions. The reported evaluation on 4,128 samples gives an overall accuracy of 98.18%, with 98.06% for
cardiovascular-risk prediction and 98.30% for diabetes-risk prediction. However, the clinical variables in the present implementation are generated from retinal-disease severity information using medically informed distributions; therefore, the reported results should be interpreted as an experimental proof of concept and require validation with real patient-level clinical measurements before clinical use.},
keywords = {multimodal learning, deep learning, retinal fundus images, EfficientNet-B3, Transformer Encoder, cardiovascular risk, diabetes risk, Grad-CAM, explainable AI},
month = {September},
}
Submit your research paper and those of your network (friends, colleagues, or peers) through your IPN account, and receive 800 INR for each paper that gets published.
Join NowNational Conference on Sustainable Engineering and Management - 2024 Last Date: 15th March 2024
Submit inquiry