Copyright © 2026 Authors retain the copyright of this article. This article is an open access article distributed under the Creative Commons Attribution License which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.
@article{208668,
author = {Rithik Raj Vaishya and Dharmin Kikaganesh and Heeya Doshi},
title = {Bridging the Perception Gap: A Multimodal AI Assistant for RAG-Driven Storytelling and Multisensory Learning},
journal = {International Journal of Innovative Research in Technology},
year = {2026},
volume = {13},
number = {4},
pages = {2342-2348},
issn = {2349-6002},
url = {https://ijirt.org/article?manuscript=208668},
abstract = {Despite the rapid increase of digital EdTech platforms and learning solutions, a significant research gap remains in the development of integrated, multimodal learning environments that move beyond static textbooks, question papers and notes to facilitate a more holistic engagement with academic material. While traditional EdTech relies on unimodal delivery, audio-visual-textual combinations are essential for students with diverse information processing styles, particularly visual, auditory, and kinesthetic learners (Fitria, 2021; Vakalopoulou et al., 2024). This paper presents a generative framework for a Multimodal AI Assistant designed to bridge this gap by transforming raw syllabus data into immersive narratives, mnemonic poems, and structured visual aids. Building on a conceptual analysis of AI-powered adaptive platforms (Liu et al., 2022; Terzopoulos & Satratzemi, 2020), the proposed system utilizes a Retrieval-Augmented Generation (RAG) architecture. The methodology outlines a semantic chunking layer along with a FAISS-based vector database for context retrieval, utilizing the Groq API (Llama 3.1) for high-speed inference to transform technical text into creative formats. The system also incorporates an immersive text-to-speech (TTS) pipeline for auditory storytelling and an automated visual synthesis module for generating presentations and flashcards. While challenges such as computational constraints and data confidentiality remain prevalent in intelligent tutoring systems (Kim et al., 2020; van Dijk, 2021), we argue that this multimodal approach has strong potential to improve information retention and learner accessibility. We conclude that generative systems offer a transformative medium for personalized education, providing a scalable model for future applications and work in modern EdTech environments.},
keywords = {Digital Edtech Platforms, unimodal information delivery, Multimodal AI, personalised learning experiences, diverse information processing styles, immersive narratives, mnemonic poems, visual aids, Retrieval Augmented Generation (RAG), semantic chunking, FAISS-based vector database, context retrieval, Groq API, llama models, Voice Processing, text-to-speech pipelines, computational constraints, data confidentiality, learner accessibility.},
month = {September},
}
Submit your research paper and those of your network (friends, colleagues, or peers) through your IPN account, and receive 800 INR for each paper that gets published.
Join NowNational Conference on Sustainable Engineering and Management - 2024 Last Date: 15th March 2024
Submit inquiry