Copyright © 2026 Authors retain the copyright of this article. This article is an open access article distributed under the Creative Commons Attribution License which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.
@article{204282,
author = {Aryan Shinde and Trupti Takale and Sneha Sul and Prof.Swati Jadhav},
title = {Multimodal Communication System for the Deaf and Speech Impaired},
journal = {International Journal of Innovative Research in Technology},
year = {2026},
volume = {13},
number = {1},
pages = {4778-4786},
issn = {2349-6002},
url = {https://ijirt.org/article?manuscript=204282},
abstract = {Communication remains a major barrier for individuals who are deaf, mute, or speech-impaired. Traditional assistive systems often handle either sign language or speech in isolation, struggle with noisy real-world conditions, and rarely produce structured machine-understandable output. This paper presents a Multimodal Voice and Sign Language to Intent Translation System that converts (a) static Indian Sign Language (ISL) gestures, (b) dynamic ISL word and sentence signs, and (c) spoken or dysarthric speech into a unified six-class intent taxonomy: REQUEST, QUESTION, STATEMENT, COMMAND, EMOTION, and UNKNOWN.
The vision pipeline uses a three-tier gesture cascade — landmark geometry rules, a MobileNetV2 convolutional neural network (CNN) for static signs, and a long short-term memory (LSTM) network for dynamic signs — combined with a pose-based 1D-CNN for full ISL sentence recognition. The audio pipeline uses faster-whisper automatic speech recognition with a dysarthric text cleaner. Multimodal fusion arbitrates between vision and speech using confidence agreement with a twelve-second pending-voice window. The complete system is deployed across three frontends — Streamlit, Flask, and OpenCV — sharing one SQLite log. Experimental results show 81.3% validation accuracy for dynamic gestures, 47.4% top-1 (75.9% top-5) for sentence recognition, and 93.75% for the optional INCLUDE-50 Transformer, with real-time inference on commodity hardware. The proposed framework offers a practical, contactless, and inclusive communication aid for the deaf and speech-impaired community.},
keywords = {Convolutional Neural Network, Dysarthric Speech, Indian Sign Language, Intent Classification, LSTM, MediaPipe, Multimodal Fusion, Pose Estimation.},
month = {June},
}
Submit your research paper and those of your network (friends, colleagues, or peers) through your IPN account, and receive 800 INR for each paper that gets published.
Join NowNational Conference on Sustainable Engineering and Management - 2024 Last Date: 15th March 2024
Submit inquiry