@inproceedings{2040bb6d8bb0413b94ee25dca3e117b6,
title = "Performance evaluation and tuning for MapReduce computing in Hadoop distributed file system",
abstract = "This paper proposes a method to facilitate the identification process for a set of configuration parameters to achieve the optimal performance with respect to a benchmark program in HDFS in an automated manner. Performance optimization of Hadoop processes is a tedious yet challenging problem due to the complexity of the systems organization with an extensive list of configuration parameters to be considered. An Automated Benchmarking Configuration Method (ABCM) is developed in this work to facilitate the identification process for the set of configuration parameters that minimizes the execution time of a benchmark, namely TestDFSIO Write and Read in particular. A two-phased configuration parameters selection process with a simple sampling technique is proposed in order to mediate the exponential computation time otherwise. By using the proposed technique, we have automatically found the sets of top five selected optimal configuration parameters that reduced the average execution time by 32% compared to the execution time with the default set of Hadoop configuration parameters.",
keywords = "benchmarks, Hadoop, Hadoop configuration, HDFS, performance tuning",
author = "Jongyeop Kim and Kumar, {T. K.Ashwin} and George, {K. M.} and Nohpill Park",
note = "Publisher Copyright: {\textcopyright} 2015 IEEE.; 13th International Conference on Industrial Informatics, INDIN 2015 ; Conference date: 22-07-2015 Through 24-07-2015",
year = "2015",
month = sep,
day = "28",
doi = "10.1109/INDIN.2015.7281711",
language = "English",
series = "Proceeding - 2015 IEEE International Conference on Industrial Informatics, INDIN 2015",
publisher = "Institute of Electrical and Electronics Engineers Inc.",
pages = "62--68",
booktitle = "Proceeding - 2015 IEEE International Conference on Industrial Informatics, INDIN 2015",
address = "United States",
}