Copyright © 2026 Authors retain the copyright of this article. This article is an open access article distributed under the Creative Commons Attribution License which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.
@article{202952,
author = {Jay Rohit Shinde and Srushti Raviraj More and Sharad Dattatray Patil and Aakash Vivekanand Solanki and Digvijay Kashinath Dongale},
title = {Autonomous Warehouse Robot Navigation Using Deep Reinforcement Learning},
journal = {International Journal of Innovative Research in Technology},
year = {2026},
volume = {12},
number = {12},
pages = {8226-8233},
issn = {2349-6002},
url = {https://ijirt.org/article?manuscript=202952},
abstract = {Autonomous navigation and task execution in warehouse environments represents a critical and growing challenge at the intersection of robotics and artificial intelligence. This paper presents the design, implementation, and comparative evaluation of an autonomous warehouse robot navigation system trained using two deep reinforcement learning algorithms: Deep Q-Network (DQN) and Proximal Policy Optimization (PPO). The system operates across two custom Gymnasium-compatible environments of differing complexity a discrete 8x8 grid employing DQN, and a larger 12x12 grid employing PPO both requiring the agent to perform a complete two-stage pick-and-place task: item retrieval followed by goal delivery, in the presence of structured shelf obstacles. A carefully engineered reward function incorporating pickup incentives, step penalties, obstacle penalties, and Manhattan-distance shaping guide the learning process. The DQN agent achieves 100% task completion across 10 item positions in a mean of 14.7 steps per episode, with a converged mean episode reward of 356, following convergence at approximately 180,000 timesteps. The PPO agent, trained across 8 parallel vectorized environments, demonstrates stable multi-location generalization across 20 item positions in the more complex 12x12 environment. A central contribution of this work is the systematic identification and resolution of four critical training failure modes task-stage ignorance, agent oscillation, collision state corruption, and catastrophic forgetting each of which independently prevents task completion. Full implementation details, comparative hyperparameter analysis, and a structured discussion of algorithmic suitability across environment scales are provided, establishing a reproducible baseline for reinforcement learning-based warehouse automation research.},
keywords = {Deep Reinforcement Learning; Deep Q-Network; Proximal Policy Optimization; Warehouse Automation; Pick-and-Place; Reward Shaping; Gymnasium; Stable-Baselines3; Autonomous Navigation; Sequential Task Learning},
month = {May},
}
Submit your research paper and those of your network (friends, colleagues, or peers) through your IPN account, and receive 800 INR for each paper that gets published.
Join NowNational Conference on Sustainable Engineering and Management - 2024 Last Date: 15th March 2024
Submit inquiry