Copyright © 2026 Authors retain the copyright of this article. This article is an open access article distributed under the Creative Commons Attribution License which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.
@article{206578,
author = {Sayali Gund},
title = {Deep Learning-Based Hand Gesture Recognition for Sign Communication},
journal = {International Journal of Innovative Research in Technology},
year = {2026},
volume = {13},
number = {2},
pages = {2265-2272},
issn = {2349-6002},
url = {https://ijirt.org/article?manuscript=206578},
abstract = {Hand gesture recognition for sign communication constitutes a critical component of next-generation assistive technologies, yet remains fundamentally constrained by challenges such as high intra-class variability, temporal ambiguity, and sensitivity to environmental perturbations. This paper presents a novel, end-to-end deep learning framework that leverages multimodal fusion and attention-driven transformer architectures for robust and real-time interpretation of sign gestures.
A custom-designed dataset, termed SGR-32, is constructed, comprising synchronized RGB video sequences and depth-informed spatial cues collected from diverse participants across unconstrained environments. To effectively capture both spatial and temporal dependencies, the proposed architecture integrates a hierarchical Convolutional Neural Network (CNN) encoder with a Vision Transformer (ViT)-inspired module, enabling global context modeling beyond local receptive fields. A temporal transformer encoder is further employed to model sequential gesture dynamics, replacing traditional recurrent units with self-attention mechanisms that provide superior long-range dependency learning.
To enhance discriminative feature learning, a multi-head self-attention mechanism is incorporated, allowing the model to focus on salient gesture regions while suppressing background noise. Additionally, a multimodal fusion strategy is designed using intermediate feature concatenation and cross-attention to effectively combine spatial (RGB), structural (depth), and motion-based representations. The training process is optimized using a hybrid loss formulation integrating focal loss and label smoothing to address class imbalance and improve generalization.
Extensive experimental evaluation demonstrates that the proposed framework achieves state-of-the-art performance, with an accuracy exceeding 97% on the SGR-32 dataset and strong robustness under real-world conditions, including occlusions, illumination changes, and dynamic backgrounds. Ablation studies further validate the contribution of transformer-based attention and multimodal fusion in enhancing recognition accuracy and stability.
The proposed system establishes a scalable and deployment-ready paradigm for intelligent sign communication, with broader implications for inclusive human–computer interaction, edge-AI assistive devices, and real-time multimodal perception systems.},
keywords = {Hand Gesture Recognition, Multimodal Fusion, Vision Transformer, Attention Mechanisms, Sign Language Interpretation, Spatiotemporal Modeling, Assistive Technology},
month = {July},
}
Submit your research paper and those of your network (friends, colleagues, or peers) through your IPN account, and receive 800 INR for each paper that gets published.
Join NowNational Conference on Sustainable Engineering and Management - 2024 Last Date: 15th March 2024
Submit inquiry