Copyright © 2026 Authors retain the copyright of this article. This article is an open access article distributed under the Creative Commons Attribution License which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.
@article{186029,
author = {Arun Karthick C and Skanath Kumar P S and Sallagarige Rahul and A. Nagaraja Rao},
title = {Performance Comparison Of Hadoop-Spark Cluster For Performing Sentiment Analysis On Big Data},
journal = {International Journal of Innovative Research in Technology},
year = {2026},
volume = {12},
number = {11},
pages = {9215-9225},
issn = {2349-6002},
url = {https://ijirt.org/article?manuscript=186029},
abstract = {The explosive growth of user-generated content on the web has driven the need for efficient sentiment analysis at a massive scale, demanding robust distributed computation. Recent advancements in big data technologies especially Apache Hadoop and Spark enable parallel processing and rapid anal- ysis of voluminous text datasets. In this study, we configure, evaluate and benchmark single-node and multi-node cluster deployments leveraging batch-oriented big data processing to classify sentiment in large movie review corpora and compare the execution time from single node versus Multi Node cluster. Our approach centers on the configuration, orchestration, and comparative analysis of different cluster setups using Hadoop and Spark as distributed batch frameworks. Experiments examine effects on performance, data locality, and scalability, providing practical insights for real-world system architects. Our results show that while cluster overhead may negate parallel speedup for small workloads, multi-node deployments consistently outper- form single-node setups when processing large-scale data. From our observations we were able to interpret that the Multi Node cluster performed 30% more efficiently than the Single Node. The study also investigates system fault tolerance during node failures and findings provide practical insights on cluster configuration and workload suitability for scalable sentiment analysis.},
keywords = {Hadoop, HDFS, Map-Reduce, Spark, Single- Node, Multi-Node Cluster, Distributed Storage, Parallel Comput- ing, Namenode, Datanode, Sentiment Analysis, Fault Tolerance, Big Data Analytics, Cluster Configuration, Sentiment Analysis.},
month = {April},
}
Submit your research paper and those of your network (friends, colleagues, or peers) through your IPN account, and receive 800 INR for each paper that gets published.
Join NowNational Conference on Sustainable Engineering and Management - 2024 Last Date: 15th March 2024
Submit inquiry