Copyright © 2026 Authors retain the copyright of this article. This article is an open access article distributed under the Creative Commons Attribution License which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.
@article{204856,
author = {Rauki Yadav and Dr. Vineet Kumar Goel and Dr. Sohit Agarwal},
title = {Advances in Deep Learning Image Captioning and Their Application for Caption-Guided Sentiment-Aware Hybrid Transformer},
journal = {International Journal of Innovative Research in Technology},
year = {2026},
volume = {13},
number = {1},
pages = {4701-4716},
issn = {2349-6002},
url = {https://ijirt.org/article?manuscript=204856},
abstract = {The generation of image captions has emerged as a promising domain of research that merges the field of visual understanding with natural language processing. The challenge of generating an image caption requires the model in question to identify the most salient objects in an image, evaluate the relationship among the items, and translate that evaluation into coherent language. The proposed work will assess a subset of the deep learning approaches that have contributed to advances in image caption generation, including early convolutional and recurrent networks, cutting-edge attention-based decoder methods, and transformer-based networks. It expands to assess recent efforts that are focused on emotional types of caption generation, fusion of multiple modalities, optimizing caption generation with reinforced learning, hybrid encoders, and transformer-based methods. Most current models have shown good performance across benchmarks, including MS COCO. However, issues concerning accurate representations of abstract scenes, the hallucination of content, and alternate application contexts persist. This paper suggests an increase in significant improvement through the use of grounding, providing valid evaluation metrics, and architectures that depict semantic structure more accurately. This paper proposes a novel multimodal framework called the Caption-Guided Sentiment-Aware Hybrid Transformer-CNN model.The proposed architecture integrates semantic scene understanding, sentiment-aware reasoning, and deep visual feature learning within a unified framework. The model is composed of three complementary modules.1.Vision-Language Model (VLM) backbone, inspired by architectures such as BLIP-2 and LLaVA, to generate scene-descriptive natural language captions from sampled classroom video frames.2.RoBERTa-based sentiment analysis network fine-tuned on educational discourse corpora, including classroom conversations, student feedback, and instructor-student interactions.3.EfficientNet-B4 convolutional neural network used as the primary visual feature extractor. The proposed framework establishes that combining semantic caption understanding, sentiment-aware contextual reasoning, and transformer-based multimodal fusion can significantly enhance automatic classroom engagement recognition, particularly in real-world and visually ambiguous learning environments.},
keywords = {Image Caption Generation, Deep Learning, Attention Mechanisms, Vision Transformers, Multimodal Learning},
month = {June},
}
Submit your research paper and those of your network (friends, colleagues, or peers) through your IPN account, and receive 800 INR for each paper that gets published.
Join NowNational Conference on Sustainable Engineering and Management - 2024 Last Date: 15th March 2024
Submit inquiry