Copyright © 2026 Authors retain the copyright of this article. This article is an open access article distributed under the Creative Commons Attribution License which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.
@article{208556,
author = {Uday Wahile and Bhagyashali Kokode and Sumit Dhale and Gaurav Thakre and Ayush Jadhav and Chinmay Thakur},
title = {VocalSense: Mapping Acoustic Patterns to Human Emotions for Intelligent Speech Interaction},
journal = {International Journal of Innovative Research in Technology},
year = {2026},
volume = {13},
number = {4},
pages = {1950-1960},
issn = {2349-6002},
url = {https://ijirt.org/article?manuscript=208556},
abstract = {Speech Emotion Recognition (SER) sits at an active crossroads of artificial intelligence research, concerned with inferring a speaker's emotional state from acoustic cues — pitch, intensity, frequency, rhythm and spectral content — rather than from the words themselves. As natural, intelligent human–computer interaction becomes more of an expectation than a novelty, interest in recognizing emotion from live speech has grown accordingly. This paper reviews recent machine-learning and deep-learning approaches to SER, with particular attention to what real-time processing demands of them. It walks through commonly used datasets such as RAVDESS and MELD, the preprocessing and feature-extraction steps typically applied to raw audio, and the central role played by Mel-Frequency Cepstral Coefficients (MFCC) as a compact way of representing speech for emotion classification. Both classical algorithms — SVM, Decision Tree, Random Forest, and Multi-Layer Perceptron — and deep architectures such as CNNs and LSTM networks are examined and weighed against one another on accuracy, computational cost, and fitness for low-latency deployment. The review finds that deep models tend to push accuracy higher while lighter architectures remain the more practical choice when latency matters, and it closes by discussing persistent challenges — dataset quality, compute cost, multilingual speech, background noise, and real-time deployment — alongside directions future work could take.},
keywords = {Speech Emotion Recognition, Machine Learning, Deep Learning, MFCC, Real-Time Emotion Detection etc.},
month = {September},
}
Submit your research paper and those of your network (friends, colleagues, or peers) through your IPN account, and receive 800 INR for each paper that gets published.
Join NowNational Conference on Sustainable Engineering and Management - 2024 Last Date: 15th March 2024
Submit inquiry