{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,27]],"date-time":"2025-03-27T11:35:08Z","timestamp":1743075308668,"version":"3.37.3"},"reference-count":37,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2022,9,5]],"date-time":"2022-09-05T00:00:00Z","timestamp":1662336000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,9,5]],"date-time":"2022-09-05T00:00:00Z","timestamp":1662336000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Cluster Comput"],"published-print":{"date-parts":[[2023,6]]},"DOI":"10.1007\/s10586-022-03728-7","type":"journal-article","created":{"date-parts":[[2022,9,5]],"date-time":"2022-09-05T15:02:55Z","timestamp":1662390175000},"page":"1891-1915","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["An experimental and comparative benchmark study examining resource utilization in managed Hadoop context"],"prefix":"10.1007","volume":"26","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7715-4652","authenticated-orcid":false,"given":"Uluer Emre","family":"\u00d6zdil","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2016-4443","authenticated-orcid":false,"given":"Serkan","family":"Ayvaz","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,9,5]]},"reference":[{"key":"3728_CR1","unstructured":"Apache Hadoop. https:\/\/hadoop.apache.org\/. Accessed 22 May 2022"},{"key":"3728_CR2","unstructured":"Announcing Amazon Elastic Compute Cloud (Amazon EC2)\u2014beta. https:\/\/aws.amazon.com\/about-aws\/whats-new\/2006\/08\/24\/announcing-amazon-elastic-compute-cloud-amazon-ec2---beta\/. Accessed 22 May 2022"},{"key":"3728_CR3","unstructured":"TPC-History. http:\/\/tpc.org\/information\/about\/history5.asp. Accessed 22 May 2022"},{"key":"3728_CR4","unstructured":"SPEC\u2014Standard Performance Evaluation Corporation. https:\/\/www.spec.org\/. Accessed 22 May 2022"},{"key":"3728_CR5","doi-asserted-by":"publisher","first-page":"580","DOI":"10.1109\/TSC.2017.2730882","volume":"11","author":"R Han","year":"2018","unstructured":"Han, R., John, L.K., Zhan, J.: Benchmarking Big Data systems: a review. IEEE Trans. Serv. Comput. 11, 580\u2013597 (2018). https:\/\/doi.org\/10.1109\/TSC.2017.2730882","journal-title":"IEEE Trans. Serv. Comput."},{"key":"3728_CR6","doi-asserted-by":"publisher","unstructured":"Ghemawat, S., Gobioff, H., Leung, S.-T.: The Google file system. In: Proceedings of the Nineteenth ACM Symposium on Operating Systems Principles, pp. 29\u201343 (2003). https:\/\/doi.org\/10.1145\/1165389.945450","DOI":"10.1145\/1165389.945450"},{"key":"3728_CR7","volume-title":"Hadoop: The Definitive Guide","author":"T White","year":"2015","unstructured":"White, T.: Hadoop: The Definitive Guide. O\u2019Reilly, Beijing (2015)"},{"key":"3728_CR8","unstructured":"Dean, J., Ghemawat, S.: MapReduce: simplified data processing on large clusters. Presented at the OSDI 2004\u20146th Symposium on Operating Systems Design and Implementation (2004)"},{"key":"3728_CR9","unstructured":"Sch\u00e4tzle, T.H., Przyjaciel-Zablocki, M., Alexander: Giant Data: MapReduce and Hadoop, ADMIN Magazine. http:\/\/www.admin-magazine.com\/HPC\/Articles\/MapReduce-and-Hadoop\/. Accessed 30 Oct 2020"},{"key":"3728_CR10","unstructured":"Ramel, B.D.: 08\/04\/2021: what are Gartner\u2019s \u201cCautions\u201d about big 3 cloud providers? https:\/\/virtualizationreview.com\/articles\/2021\/08\/04\/gartner-cloud-2021.aspx. Accessed 15 Apr 2021"},{"key":"3728_CR11","unstructured":"Azure HDInsight\u2014Hadoop, Spark, & Kafka Service\u2014Microsoft Azure. https:\/\/azure.microsoft.com\/en-us\/services\/hdinsight\/. Accessed 8 Jan 2021"},{"key":"3728_CR12","unstructured":"Announcing general availability of Azure HDInsight 3.6. https:\/\/azure.microsoft.com\/en-us\/blog\/announcing-general-availability-of-azure-hdinsight-3-6\/. Accessed 14 Jan 2021"},{"key":"3728_CR13","unstructured":"Dataproc. https:\/\/cloud.google.com\/dataproc. Accessed 8 Jan 2021"},{"key":"3728_CR14","unstructured":"Compute Engine: Virtual Machines (VMs). https:\/\/cloud.google.com\/compute. Accessed 8 Jan 2021"},{"key":"3728_CR15","unstructured":"What is E-MapReduce?\u2014Product Introduction\u2014Alibaba Cloud Documentation Center. https:\/\/www.alibabacloud.com\/help\/doc-detail\/28068.htm?spm=a2c63.l28256.b99.4.65e270b2YXyKDV. Accessed 14 Jan 2021"},{"key":"3728_CR16","unstructured":"Elastic Compute Service (ECS): Elastic & Secure Cloud Servers\u2014Alibaba Cloud. https:\/\/www.alibabacloud.com\/product\/ecs. Accessed 17 Jan 2021"},{"key":"3728_CR17","unstructured":"Alibaba Cloud Linux OS. https:\/\/alibaba.github.io\/cloud-kernel\/os.html. Accessed 14 Jan 2021"},{"key":"3728_CR18","doi-asserted-by":"publisher","unstructured":"Huang, S., Huang, J., Dai, J., Xie, T., Huang, B.: The HiBench benchmark suite: characterization of the MapReduce-based data analysis. In: 2010 IEEE 26th International Conference on Data Engineering Workshops (ICDEW 2010), pp. 41\u201351 (2010). https:\/\/doi.org\/10.1109\/ICDEW.2010.5452747","DOI":"10.1109\/ICDEW.2010.5452747"},{"key":"3728_CR19","unstructured":"GitHub\u2014Intel-bigdata\/HiBench. HiBench is a big data benchmark suite. https:\/\/github.com\/Intel-bigdata\/HiBench. Accessed 8 Jan 2021"},{"key":"3728_CR20","doi-asserted-by":"publisher","first-page":"100329","DOI":"10.1016\/j.bdr.2022.100329","volume":"29","author":"Q Guo","year":"2022","unstructured":"Guo, Q., Xie, Y., Li, Q., Zhu, Y.: XDataExplorer: a three-stage comprehensive self-tuning tool for Big Data platforms. Big Data Res. 29, 100329 (2022). https:\/\/doi.org\/10.1016\/j.bdr.2022.100329","journal-title":"Big Data Res."},{"key":"3728_CR21","doi-asserted-by":"publisher","first-page":"100186","DOI":"10.1016\/j.bdr.2021.100186","volume":"24","author":"L Sfaxi","year":"2021","unstructured":"Sfaxi, L., Aissa, M.M.B.: Babel: a generic benchmarking platform for Big Data architectures. Big Data Res. 24, 100186 (2021)","journal-title":"Big Data Res."},{"issue":"12","key":"3728_CR22","doi-asserted-by":"publisher","first-page":"2983","DOI":"10.1109\/TPDS.2021.3080702","volume":"32","author":"P Prieto","year":"2021","unstructured":"Prieto, P., Abad, P., Gregorio, J.A., Puente, V.: Fast, accurate processor evaluation through heterogeneous, sample-based benchmarking. IEEE Trans. Parallel Distrib. Syst. 32(12), 2983\u20132995 (2021)","journal-title":"IEEE Trans. Parallel Distrib. Syst."},{"issue":"4","key":"3728_CR23","doi-asserted-by":"publisher","first-page":"3381","DOI":"10.1007\/s10586-021-03339-8","volume":"24","author":"R Ghazali","year":"2021","unstructured":"Ghazali, R., Adabi, S., Down, D.G., Movaghar, A.: A classification of Hadoop job schedulers based on performance optimization approaches. Clust. Comput. 24(4), 3381\u20133403 (2021)","journal-title":"Clust. Comput."},{"key":"3728_CR24","doi-asserted-by":"publisher","first-page":"1035","DOI":"10.1007\/s10586-021-03512-z","volume":"25","author":"R Ghafari","year":"2022","unstructured":"Ghafari, R., Kabutarkhani, F.H., Mansouri, N.: Task scheduling algorithms for energy optimization in cloud environment: a comprehensive review. Clust. Comput. 25, 1035\u20131093 (2022). https:\/\/doi.org\/10.1007\/s10586-021-03512-z","journal-title":"Clust. Comput."},{"key":"3728_CR25","doi-asserted-by":"publisher","DOI":"10.1109\/TCC.2021.3108043","author":"D Cheng","year":"2021","unstructured":"Cheng, D., Wang, Y., Dai, D.: Dynamic resource provisioning for iterative workloads on Apache Spark. IEEE Trans. Cloud Comput. (2021). https:\/\/doi.org\/10.1109\/TCC.2021.3108043","journal-title":"IEEE Trans. Cloud Comput."},{"issue":"2","key":"3728_CR26","doi-asserted-by":"publisher","first-page":"1421","DOI":"10.1007\/s10586-022-03541-2","volume":"25","author":"C Li","year":"2022","unstructured":"Li, C., Cai, Q., Luo, Y.: Dynamic data replacement and adaptive scheduling policies in spark. Clust. Comput. 25(2), 1421\u20131439 (2022). https:\/\/doi.org\/10.1007\/s10586-022-03541-2","journal-title":"Clust. Comput."},{"key":"3728_CR27","doi-asserted-by":"publisher","first-page":"100206","DOI":"10.1016\/j.bdr.2021.100206","volume":"25","author":"RLDC Costa","year":"2021","unstructured":"Costa, R.L.D.C., Moreira, J., Pintor, P., dos Santos, V., Lifschitz, S.: A survey on data-driven performance tuning for big data analytics platforms. Big Data Res. 25, 100206 (2021)","journal-title":"Big Data Res."},{"key":"3728_CR28","doi-asserted-by":"publisher","unstructured":"Poggi, N., Montero, A., Carrera, D.: Characterizing BigBench queries, hive, and spark in multi-cloud environments. Lecture Notes in Computer Science (Including Subseries Lecture Notes in Artificial Intelligence and Lecture Notes in Bioinformatics). 10661 LNCS, pp. 55\u201374 (2018). https:\/\/doi.org\/10.1007\/978-3-319-72401-0_5","DOI":"10.1007\/978-3-319-72401-0_5"},{"key":"3728_CR29","doi-asserted-by":"publisher","unstructured":"Wang, H., Shen, H., Reiss, C., Jain, A., Zhang, Y.: Improved intermediate data management for MapReduce frameworks. Presented at the Proceedings\u20142020 IEEE 34th International Parallel and Distributed Processing Symposium, IPDPS 2020 (2020). https:\/\/doi.org\/10.1109\/IPDPS47924.2020.00062","DOI":"10.1109\/IPDPS47924.2020.00062"},{"key":"3728_CR30","doi-asserted-by":"publisher","first-page":"130","DOI":"10.1109\/TPDS.2015.2398438","volume":"27","author":"K Hwang","year":"2016","unstructured":"Hwang, K., Bai, X., Shi, Y., Li, M., Chen, W.-G., Wu, Y.: Cloud performance modeling with benchmark evaluation of elastic scaling strategies. IEEE Trans. Parallel Distrib. Syst. 27, 130\u2013143 (2016). https:\/\/doi.org\/10.1109\/TPDS.2015.2398438","journal-title":"IEEE Trans. Parallel Distrib. Syst."},{"key":"3728_CR31","doi-asserted-by":"publisher","unstructured":"Ahn, H., Kim, H., You, W.: Performance study of spark on YARN cluster using HiBench. Presented at the 2018 IEEE International Conference on Consumer Electronics\u2014Asia, ICCE-Asia 2018 (2018). https:\/\/doi.org\/10.1109\/ICCE-ASIA.2018.8552137","DOI":"10.1109\/ICCE-ASIA.2018.8552137"},{"key":"3728_CR32","doi-asserted-by":"publisher","unstructured":"Han, S., Choi, W., Muwafiq, R., Nah, Y.: Impact of memory size on bigdata processing based on Hadoop and Spark. Presented at the Proceedings of the 2017 Research in Adaptive and Convergent Systems, RACS 2017 (2017). https:\/\/doi.org\/10.1145\/3129676.3129688","DOI":"10.1145\/3129676.3129688"},{"key":"3728_CR33","doi-asserted-by":"publisher","DOI":"10.1002\/cpe.4367","author":"Y Samadi","year":"2018","unstructured":"Samadi, Y., Zbakh, M., Tadonki, C.: Performance comparison between Hadoop and spark frameworks using HiBench benchmarks. Concurr. Comput. (2018). https:\/\/doi.org\/10.1002\/cpe.4367","journal-title":"Concurr. Comput."},{"issue":"1","key":"3728_CR34","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1186\/s40537-021-00499-7","volume":"8","author":"N Ahmed","year":"2021","unstructured":"Ahmed, N., Barczak, A.L., Rashid, M.A., Susnjak, T.: A parallelization model for performance characterization of Spark Big Data jobs on Hadoop clusters. J. Big Data 8(1), 1\u201328 (2021)","journal-title":"J. Big Data"},{"key":"3728_CR35","doi-asserted-by":"publisher","unstructured":"Shih, W.C., Yang, C.T., Ranjan, R., Chiang, C.I: Implementation and evaluation of a container management platform on Docker: Hadoop deployment as an example. Clust. Comput. 24(4), 3421\u20133430 (2021). https:\/\/doi.org\/10.1007\/s10586-021-03337-w","DOI":"10.1007\/s10586-021-03337-w"},{"key":"3728_CR36","unstructured":"GitHub Repository of the study. https:\/\/github.com\/emretto\/benchmark-hadoop-on-paas. Accessed 24 May 2022"},{"key":"3728_CR37","unstructured":"Jota juliojsb\/sarviewer. https:\/\/github.com\/juliojsb\/sarviewer. Accessed 12 Dec 2020"}],"container-title":["Cluster Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10586-022-03728-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10586-022-03728-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10586-022-03728-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,5,23]],"date-time":"2023-05-23T19:12:40Z","timestamp":1684869160000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10586-022-03728-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,9,5]]},"references-count":37,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2023,6]]}},"alternative-id":["3728"],"URL":"https:\/\/doi.org\/10.1007\/s10586-022-03728-7","relation":{},"ISSN":["1386-7857","1573-7543"],"issn-type":[{"type":"print","value":"1386-7857"},{"type":"electronic","value":"1573-7543"}],"subject":[],"published":{"date-parts":[[2022,9,5]]},"assertion":[{"value":"23 December 2021","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"15 August 2022","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 August 2022","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 September 2022","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"Author Serkan Ayvaz and Author Uluer Emre Ozdil declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"The authors consciously assure that this material is the authors\u2019 own original work, which is not currently being considered for publication elsewhere. This article does not contain any studies with human participants or animals performed by any of the authors.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical approval"}},{"value":"This article does not contain any studies with human participants or animals performed by any of the authors. The consent is not a requirement for this study.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Informed consent"}}]}}