{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,26]],"date-time":"2025-03-26T08:02:31Z","timestamp":1742976151408,"version":"3.40.3"},"publisher-location":"Cham","reference-count":36,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783319699523"},{"type":"electronic","value":"9783319699530"}],"license":[{"start":{"date-parts":[[2018,1,1]],"date-time":"2018-01-01T00:00:00Z","timestamp":1514764800000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by\/4.0"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2018]]},"DOI":"10.1007\/978-3-319-69953-0_5","type":"book-chapter","created":{"date-parts":[[2018,3,19]],"date-time":"2018-03-19T14:54:32Z","timestamp":1521471272000},"page":"70-89","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["On the Performance of Spark on HPC Systems: Towards a Complete Picture"],"prefix":"10.1007","author":[{"given":"Orcun","family":"Yildiz","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shadi","family":"Ibrahim","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2018,3,20]]},"reference":[{"key":"5_CR1","unstructured":"Big Data and Extreme-scale Computing (BDEC) Workshop. http:\/\/www.exascale.org\/bdec\/"},{"key":"5_CR2","unstructured":"HiBench Big Data microbenchmark suite. https:\/\/github.com\/intel-hadoop\/HiBench"},{"key":"5_CR3","unstructured":"The Apache Hadoop Project. http:\/\/www.hadoop.org"},{"key":"5_CR4","unstructured":"Apache Storm (2012). https:\/\/storm.apache.org\/"},{"key":"5_CR5","unstructured":"Apache Spark primer (2017). http:\/\/go.databricks.com\/hubfs\/pdfs\/Apache_Spark_Primer_170303.pdf"},{"key":"5_CR6","unstructured":"IDC\u2019s Data Age 2025 study (2017). http:\/\/www.seagate.com\/www-content\/our-story\/trends\/files\/Seagate-WP-DataAge2025-March-2017.pdf"},{"key":"5_CR7","unstructured":"Powered by Hadoop (2017). http:\/\/wiki.apache.org\/hadoop\/PoweredBy\/"},{"key":"5_CR8","unstructured":"Hadoop Workload Analysis. http:\/\/www.pdl.cmu.edu\/HLA\/index.shtml . Accessed Jan 2017"},{"key":"5_CR9","doi-asserted-by":"crossref","unstructured":"Bent, J., Faibish, S., Ahrens, J., Grider, G., Patchett, J., Tzelnic, P., Woodring, J.: Jitter-free co-processing on a prototype exascale storage stack. In: 2012 IEEE 28th Symposium on Mass Storage Systems and Technologies (MSST), pp. 1\u20135. IEEE (2012)","DOI":"10.1109\/MSST.2012.6232382"},{"issue":"1\u20132","key":"5_CR10","first-page":"285","volume":"3","author":"Y Bu","year":"2010","unstructured":"Bu, Y., Howe, B., Balazinska, M., Ernst, M.D.: Haloop: efficient iterative data processing on large clusters. Int. J. Very Large Databases 3(1\u20132), 285\u2013296 (2010)","journal-title":"Int. J. Very Large Databases"},{"issue":"4","key":"5_CR11","first-page":"28","volume":"36","author":"P Carbone","year":"2015","unstructured":"Carbone, P., Katsifodimos, A., Ewen, S., Markl, V., Haridi, S., Tzoumas, K.: Apache flink: stream and batch processing in a single engine. Bull. IEEE Comput. Soc. Tech. Comm. Data Eng. 36(4), 28\u201338 (2015)","journal-title":"Bull. IEEE Comput. Soc. Tech. Comm. Data Eng."},{"key":"5_CR12","doi-asserted-by":"crossref","unstructured":"Chaimov, N., Malony, A., Canon, S., Iancu, C., Ibrahim, K.Z., Srinivasan, J.: Scaling Spark on HPC systems. In: Proceedings of the 25th ACM International Symposium on High-Performance Parallel and Distributed Computing, pp. 97\u2013110. ACM (2016)","DOI":"10.1145\/2907294.2907310"},{"issue":"1","key":"5_CR13","doi-asserted-by":"publisher","first-page":"107","DOI":"10.1145\/1327452.1327492","volume":"51","author":"J Dean","year":"2008","unstructured":"Dean, J., Ghemawat, S.: MapReduce: simplified data processing on large clusters. Commun. ACM 51(1), 107\u2013113 (2008)","journal-title":"Commun. ACM"},{"key":"5_CR14","unstructured":"Donovan, S., Huizenga, G., Hutton, A.J., Ross, C.C., Petersen, M.K., Schwan, P.: Lustre: building a file system for 1000-node clusters (2003)"},{"issue":"3","key":"5_CR15","first-page":"15","volume":"3","author":"M Dorier","year":"2016","unstructured":"Dorier, M., Antoniu, G., Cappello, F., Snir, M., Sisneros, R., Yildiz, O., Ibrahim, S., Peterka, T., Orf, L.: Damaris: addressing performance variability in data management for post-petascale simulations. ACM Trans. Parallel Comput. (TOPC) 3(3), 15 (2016)","journal-title":"ACM Trans. Parallel Comput. (TOPC)"},{"key":"5_CR16","doi-asserted-by":"crossref","unstructured":"Dorier, M., Antoniu, G., Ross, R., Kimpe, D., Ibrahim, S.: CALCioM: mitigating I\/O interference in HPC systems through cross-application coordination. In: Proceedings of the IEEE International Parallel and Distributed Processing Symposium (IPDPS 2014), Phoenix, AZ, USA, May 2014. http:\/\/hal.inria.fr\/hal-00916091","DOI":"10.1109\/IPDPS.2014.27"},{"key":"5_CR17","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"3","DOI":"10.1007\/978-3-319-49748-8_1","volume-title":"Big Data Benchmarking","author":"G Fox","year":"2016","unstructured":"Fox, G., Qiu, J., Jha, S., Ekanayake, S., Kamburugamuve, S.: Big data, simulations and HPC convergence. In: Rabl, T., Nambiar, R., Baru, C., Bhandarkar, M., Poess, M., Pyne, S. (eds.) WBDB -2015. LNCS, vol. 10044, pp. 3\u201317. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-49748-8_1"},{"key":"5_CR18","doi-asserted-by":"crossref","unstructured":"Gainaru, A., Aupy, G., Benoit, A., Cappello, F., Robert, Y., Snir, M.: Scheduling the I\/O of HPC applications under congestion. In: International Parallel and Distributed Processing Symposium, pp. 1013\u20131022. IEEE (2015)","DOI":"10.1109\/IPDPS.2015.116"},{"key":"5_CR19","doi-asserted-by":"crossref","unstructured":"Guo, Y., Bland, W., Balaji, P., Zhou, X.: Fault tolerant MapReduce-MPI for HPC clusters. In: Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis, p. 34. ACM (2015)","DOI":"10.1145\/2807591.2807617"},{"issue":"3","key":"5_CR20","doi-asserted-by":"publisher","first-page":"59","DOI":"10.1145\/1272998.1273005","volume":"41","author":"Michael Isard","year":"2007","unstructured":"Isard, M., Budiu, M., Yu, Y., Birrell, A., Fetterly, D.: Dryad: distributed data-parallel programs from sequential building blocks. In: Special Interest Group on Operating Systems Review, vol. 41, pp. 59\u201372. ACM (2007)","journal-title":"ACM SIGOPS Operating Systems Review"},{"key":"5_CR21","doi-asserted-by":"crossref","unstructured":"Islam, N.S., Wasi-ur Rahman, M., Lu, X., Panda, D.K.: High performance design for HDFS with byte-addressability of NVM and RDMA. In: Proceedings of the 2016 International Conference on Supercomputing, p. 8. ACM (2016)","DOI":"10.1145\/2925426.2926290"},{"issue":"4","key":"5_CR22","doi-asserted-by":"publisher","first-page":"481","DOI":"10.1177\/1094342006070078","volume":"20","author":"Y J\u00e9gou","year":"2006","unstructured":"J\u00e9gou, Y., Lant\u00e9ri, S., Leduc, J., Melab, N., Mornet, G., Namyst, R., Primet, P., Quetier, B., Richard, O., Talbi, E.G., Ir\u00e9a, T.: Grid\u20195000: a large scale and highly reconfigurable experimental grid testbed. Int. J. High Perform. Comput. Appl. 20(4), 481\u2013494 (2006)","journal-title":"Int. J. High Perform. Comput. Appl."},{"key":"5_CR23","doi-asserted-by":"publisher","first-page":"373","DOI":"10.1002\/9780470940105.ch14","volume-title":"Cloud Computing: Principles and Paradigms","author":"H Jin","year":"2011","unstructured":"Jin, H., Ibrahim, S., Qi, L., Cao, H., Wu, S., Shi, X.: The MapReduce programming model and implementations. In: Buyya, R., Broberg, J., Goscinski, A. (eds.) Cloud Computing: Principles and Paradigms, pp. 373\u2013390. Wiley, New York (2011)"},{"key":"5_CR24","doi-asserted-by":"crossref","unstructured":"Li, Z., Shen, H.: Designing a hybrid scale-up\/out hadoop architecture based on performance measurements for high application performance. In: 2015 44th International Conference on Parallel Processing (ICPP), pp. 21\u201330. IEEE (2015)","DOI":"10.1109\/ICPP.2015.11"},{"key":"5_CR25","doi-asserted-by":"crossref","unstructured":"Lofstead, J., Zheng, F., Liu, Q., Klasky, S., Oldfield, R., Kordenbrock, T., Schwan, K., Wolf, M.: Managing variability in the I\/O performance of petascale storage systems. In: International Conference for High Performance Computing, Networking, Storage and Analysis, pp. 1\u201312. IEEE (2010)","DOI":"10.1109\/SC.2010.32"},{"key":"5_CR26","unstructured":"Lopez, I.: IDC talks convergence in high performance data analysis (2013). https:\/\/www.datanami.com\/2013\/06\/19\/idc_talks_convergence_in_high_performance_data_analysis\/"},{"key":"5_CR27","doi-asserted-by":"crossref","unstructured":"Ross, R.B., Thakur, R., et al.: PVFS: a parallel file system for Linux clusters. In: Annual Linux Showcase and Conference, pp. 391\u2013430 (2000)","DOI":"10.7551\/mitpress\/1556.003.0022"},{"key":"5_CR28","doi-asserted-by":"crossref","unstructured":"Sato, K., Mohror, K., Moody, A., Gamblin, T., de Supinski, B.R., Maruyama, N., Matsuoka, S.: A user-level infiniband-based file system and checkpoint strategy for burst buffers. In: 2014 14th IEEE\/ACM International Symposium on Cluster, Cloud and Grid Computing (CCGrid), pp. 21\u201330. IEEE (2014)","DOI":"10.1109\/CCGrid.2014.24"},{"key":"5_CR29","unstructured":"Shan, H., Shalf, J.: Using IOR to analyze the I\/O performance for HPC platforms. In: Cray User Group Conference 2007, Seattle, WA, USA (2007)"},{"key":"5_CR30","unstructured":"Tantisiriroj, W., Patil, S., Gibson, G.: Data-intensive file systems for internet services: a rose by any other name. Parallel Data Laboratory, Technical report UCB\/EECS-2008-99 (2008)"},{"key":"5_CR31","doi-asserted-by":"crossref","unstructured":"Tous, R., Gounaris, A., Tripiana, C., Torres, J., Girona, S., Ayguad\u00e9, E., Labarta, J., Becerra, Y., Carrera, D., Valero, M.: Spark deployment and performance evaluation on the MareNostrum supercomputer. In: 2015 IEEE International Conference on Big Data (Big Data), pp. 299\u2013306. IEEE (2015)","DOI":"10.1109\/BigData.2015.7363768"},{"key":"5_CR32","doi-asserted-by":"crossref","unstructured":"Wang, Y., Goldstone, R., Yu, W., Wang, T.: Characterization and optimization of memory-resident MapReduce on HPC systems. In: 2014 IEEE 28th International Parallel and Distributed Processing Symposium, pp. 799\u2013808. IEEE (2014)","DOI":"10.1109\/IPDPS.2014.87"},{"key":"5_CR33","doi-asserted-by":"crossref","unstructured":"Yildiz, O., Dorier, M., Ibrahim, S., Ross, R., Antoniu, G.: On the root causes of cross-application I\/O interference in HPC storage systems. In: IPDPS-International Parallel and Distributed Processing Symposium (2016)","DOI":"10.1109\/IPDPS.2016.50"},{"key":"5_CR34","doi-asserted-by":"crossref","unstructured":"Yildiz, O., Zhou, A.C., Ibrahim, S.: Eley: on the effectiveness of burst buffers for big data processing in HPC systems. In: 2017 IEEE International Conference on Cluster Computing (CLUSTER), pp. 87\u201391, September 2017","DOI":"10.1109\/CLUSTER.2017.73"},{"key":"5_CR35","doi-asserted-by":"crossref","unstructured":"Zaharia, M., Borthakur, D., Sen Sarma, J., Elmeleegy, K., Shenker, S., Stoica, I.: Delay scheduling: a simple technique for achieving locality and fairness in cluster scheduling. In: Proceedings of the 5th European Conference on Computer Systems, pp. 265\u2013278. ACM (2010)","DOI":"10.1145\/1755913.1755940"},{"key":"5_CR36","unstructured":"Zaharia, M., Chowdhury, M., Franklin, M.J., Shenker, S., Stoica, I.: Spark: cluster computing with working sets. In: HotCloud 2010, p. 10 (2010)"}],"container-title":["Lecture Notes in Computer Science","Supercomputing Frontiers"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-69953-0_5","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,7,2]],"date-time":"2024-07-02T03:55:49Z","timestamp":1719892549000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-319-69953-0_5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018]]},"ISBN":["9783319699523","9783319699530"],"references-count":36,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-69953-0_5","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2018]]}}}