{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,9,9]],"date-time":"2024-09-09T04:55:34Z","timestamp":1725857734934},"publisher-location":"Cham","reference-count":52,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783319413204"},{"type":"electronic","value":"9783319413211"}],"license":[{"start":{"date-parts":[[2016,1,1]],"date-time":"2016-01-01T00:00:00Z","timestamp":1451606400000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2016]]},"DOI":"10.1007\/978-3-319-41321-1_25","type":"book-chapter","created":{"date-parts":[[2016,6,14]],"date-time":"2016-06-14T06:19:15Z","timestamp":1465885155000},"page":"486-504","source":"Crossref","is-referenced-by-count":4,"title":["Multi-versioning Performance Opportunities in BGAS System for Resilience"],"prefix":"10.1007","author":[{"given":"Nan","family":"Dun","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dirk","family":"Pleiter","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Aiman","family":"Fang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Nicolas","family":"Vandenbergen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Andrew A.","family":"Chien","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2016,6,15]]},"reference":[{"key":"25_CR1","unstructured":"IOR benchmark. http:\/\/ior-sio.sourceforge.net"},{"key":"25_CR2","unstructured":"Scalable checkpoint\/restart (SCR) library. https:\/\/github.com\/hpc\/scr"},{"key":"25_CR3","unstructured":"Summit compute system. https:\/\/www.olcf.ornl.gov\/summit\/"},{"key":"25_CR4","unstructured":"Antypas, K., Wright, N., Cardo, N.P., Andrews, A., Cordery, M.: Cori: a Cray XC pre-exascale system for NERSC. In: Cray User Group Proceedings. Cray (2014)"},{"key":"25_CR5","unstructured":"Bariuso, R., Knies, A.: SHMEM user\u2019s guide for C. Cray Research, Inc. (1994)"},{"key":"25_CR6","doi-asserted-by":"crossref","unstructured":"Bautista-Gomez, L., Tsuboi, S., Komatitsch, D., Cappello, F., Maruyama, N., Matsuoka, S.: FTI: high performance fault tolerance interface for hybrid systems. In: Proceedings of the 2011 ACM\/IEEE International Conference for High Performance Computing, Networking, Storage and Analysis, SC 2011 (2011)","DOI":"10.1145\/2063384.2063427"},{"key":"25_CR7","doi-asserted-by":"crossref","unstructured":"Bent, J., Grider, G., Kettering, B., Manzanares, A., McClelland, M., Torres, A., Torrez, A.: Storage challenges at Los Alamos National Lab. In: IEEE 28th Symposium on Mass Storage Systems and Technologies, pp. 1\u20135, April 2012","DOI":"10.1109\/MSST.2012.6232376"},{"key":"25_CR8","unstructured":"Bergman, K., Borkar, S., Campbell, D., Carlson, W., Dally, W., Denneau, M., Franzon, P., Harrod, W., Hiller, J., Karp, S., Keckler, S., Klein, D., Lucas, R., Richards, M., Scarpelli, A., Scott, S., Snavely, A., Sterling, T., Williams, R.S., Yelick, K.: Exascale computing study: technology challenges in achieving exascale systems. Technical report DARPA IPTO (2008)"},{"key":"25_CR9","doi-asserted-by":"crossref","first-page":"67","DOI":"10.1145\/1941487.1941507","volume":"54","author":"S Borkar","year":"2011","unstructured":"Borkar, S., Chien, A.A.: The future of microprocessors. Commun. ACM 54, 67\u201377 (2011)","journal-title":"Commun. ACM"},{"key":"25_CR10","unstructured":"Brown, D.L., Messina, P., Keyes, D., Morrison, J., Lucas, R., Shalf, J., Beckman, P., Brightwell, R., Geist, A., Vetter, J., et al.: Scientific grand challenges: crosscutting technologies for computing at the exascale. Office of Science, U.S. Department of Energy, pp. 2\u20134, February 2010"},{"issue":"3","key":"25_CR11","doi-asserted-by":"crossref","first-page":"212","DOI":"10.1177\/1094342009106189","volume":"23","author":"F Cappello","year":"2009","unstructured":"Cappello, F.: Fault tolerance in petascale\/exascale systems: current knowledge, challenges and research opportunities. Int. J. High Perform. Comput. Appl. 23(3), 212\u2013226 (2009)","journal-title":"Int. J. High Perform. Comput. Appl."},{"issue":"02","key":"25_CR12","doi-asserted-by":"crossref","first-page":"111","DOI":"10.1142\/S0129626411000126","volume":"21","author":"F Cappello","year":"2011","unstructured":"Cappello, F., Casanova, H., Robert, Y.: Preventive migration vs. preventive checkpointing for extreme scale supercomputers. Parallel Process. Lett. 21(02), 111\u2013132 (2011)","journal-title":"Parallel Process. Lett."},{"key":"25_CR13","doi-asserted-by":"crossref","unstructured":"Carns, P., Latham, R., Ross, R., Iskra, K., Lang, S., Riley, K.: 24\/7 characterization of petascale I\/O workloads. In: IEEE International Conference on Cluster Computing and Workshops, pp. 1\u201310, August 2009","DOI":"10.1109\/CLUSTR.2009.5289150"},{"key":"25_CR14","doi-asserted-by":"crossref","unstructured":"Chien, A.A., Balaji, P., Beckman, P., Dun, N., Fang, A., Fujita, H., Iskra, K., Rubenstein, Z., Zheng, Z., Schreiber, R., Hammond, J., Dinan, J., Laguna, I., Dubey, A., Hoemmen, M., Heroux, M., Teranishi, K., Siegel, A.: Versioned distributed arrays for resilience in scientific applications: global view resilience. In: Proceedings of International Conference on Computational Science (2015)","DOI":"10.1016\/j.procs.2015.05.187"},{"key":"25_CR15","doi-asserted-by":"crossref","unstructured":"Daly, J.T.: A higher order estimate of the optimum checkpoint interval for restart dumps. Future Gener. Comput. Syst. 22(3) (2006)","DOI":"10.1016\/j.future.2004.11.016"},{"key":"25_CR16","doi-asserted-by":"crossref","unstructured":"Dong, X., Muralimanohar, N., Jouppi, N., Kaufmann, R., Xie, Y.: Leveraging 3D PCRAM technologies to reduce checkpoint overhead for future exascale systems. In: Proceedings of the Conference on High Performance Computing Networking, Storage and Analysis, SC 2009, pp. 57:1\u201357:12 (2009)","DOI":"10.1145\/1654059.1654117"},{"key":"25_CR17","doi-asserted-by":"crossref","unstructured":"Dun, N., Fujita, H., Tramm, J., Chien, A.A., Siegel, A.R.: Data decomposition in Monte Carlo particle transport simulations using global view arrays. Int. J. High Perform. Comput. Appl. March 2015","DOI":"10.1177\/1094342015577681"},{"issue":"3","key":"25_CR18","doi-asserted-by":"crossref","first-page":"1302","DOI":"10.1007\/s11227-013-0884-0","volume":"65","author":"IP Egwutuoha","year":"2013","unstructured":"Egwutuoha, I.P., Levy, D., Selic, B., Chen, S.: A survey of fault tolerance mechanisms and checkpoint\/restart implementations for high performance computing systems. J. Supercomput. 65(3), 1302\u20131326 (2013)","journal-title":"J. Supercomput."},{"key":"25_CR19","doi-asserted-by":"crossref","unstructured":"Fang, A., Chien, A.A.: How much SSD is useful for resilience in supercomputers. In: Proceedings of the 5th Workshop on Fault Tolerance for HPC at eXtreme Scale (2015)","DOI":"10.1145\/2751504.2751509"},{"key":"25_CR20","doi-asserted-by":"crossref","unstructured":"Ferreira, K., Stearley, J., Laros III, J.H., Oldfield, R., Pedretti, K., Brightwell, R., Riesen, R., Bridges, P.G., Arnold, D.: Evaluating the viability of process replication reliability for exascale systems. In: Proceedings of 2011 International Conference for High Performance Computing, Networking, Storage and Analysis (2011)","DOI":"10.1145\/2063384.2063443"},{"key":"25_CR21","doi-asserted-by":"crossref","unstructured":"Fiala, D., Mueller, F., Engelmann, C., Riesen, R., Ferreira, K., Brightwell, R.: Detection and correction of silent data corruption for large-scale high-performance computing. In: Proceedings of 2012 International Conference on High Performance Computing, Networking, Storage and Analysis, pp. 78:1\u201378:12 (2012)","DOI":"10.1109\/SC.2012.49"},{"key":"25_CR22","unstructured":"Fitch, B.G.: Exploring the capabilities of a massively scalable, compute-in-storage architecture (2013). http:\/\/www.hpdc.org\/2013\/site\/files\/HPDC13_Fitch_BlueGeneActiveStorage.pdf"},{"key":"25_CR23","doi-asserted-by":"crossref","unstructured":"Fitch, B.G., Rayshubskiy, A., Pitman, M.C., Ward, T.J.C., Germain, R.S.: Using the active storage fabrics model to address petascale storage challenges. In: Proceedings of the 4th Annual Workshop on Petascale Data Storage (2009)","DOI":"10.1145\/1713072.1713086"},{"key":"25_CR24","doi-asserted-by":"crossref","unstructured":"Fujita, H., Dun, N., Rubenstein, Z.A., Chien, A.A.: Log-structured global array for efficient multi-version snapshots. In: Proceedings of 2015 15th IEEE\/ACM International Symposium on Cluster, Cloud and Grid Computing, pp. 281\u2013291 (2015)","DOI":"10.1109\/CCGrid.2015.80"},{"key":"25_CR25","doi-asserted-by":"crossref","unstructured":"Fujita, H., Iskra, K., Balaji, P., Chien, A.A.: Empirical comparison of three versioning architectures. In: Proceedings of IEEE Cluster 2015 (2015)","DOI":"10.1109\/CLUSTER.2015.69"},{"key":"25_CR26","doi-asserted-by":"crossref","unstructured":"Gao, S., He, B., Xu, J.: Real-time in-memory checkpointing for future hybrid memory systems. In: Proceedings of the 29th ACM on International Conference on Supercomputing, pp. 263\u2013272 (2015)","DOI":"10.1145\/2751205.2751212"},{"key":"25_CR27","unstructured":"GVR Team.: Global View Resilience (GVR) API documentation, version 1.0.1. Technical report, University of Chicago, Department of Computer Science, October 2015"},{"key":"25_CR28","doi-asserted-by":"crossref","first-page":"494","DOI":"10.1088\/1742-6596\/46\/1\/067","volume":"46","author":"PH Hargrove","year":"2006","unstructured":"Hargrove, P.H., Duell, J.C.: Berkeley lab checkpoint\/restart (BLCR) for Linux clusters. J. Phys. Conf. Ser. 46, 494 (2006)","journal-title":"J. Phys. Conf. Ser."},{"key":"25_CR29","unstructured":"Heger, D., Shah, G.: IBM\u2019s general parallel file system (GPFS) 1.4 for AIX. Technical report, IBM Corporation, November 2001"},{"issue":"6","key":"25_CR30","doi-asserted-by":"crossref","first-page":"518","DOI":"10.1109\/TC.1984.1676475","volume":"C\u201333","author":"KH Huang","year":"1984","unstructured":"Huang, K.H., Abraham, J.: Algorithm-based fault tolerance for matrix operations. IEEE Trans. Comput. C\u201333(6), 518\u2013528 (1984)","journal-title":"IEEE Trans. Comput."},{"key":"25_CR31","doi-asserted-by":"crossref","unstructured":"IBM Blue Gene Team: The IBM Blue Gene project. IBM J. Res. Dev. 57 (2013)","DOI":"10.1147\/JRD.2012.2220487"},{"key":"25_CR32","doi-asserted-by":"crossref","unstructured":"Jones, T., Koniges, A., Yates, R.K.: Performance of the IBM general parallel file system. In: Proceedings of 2000 IEEE International Parallel and Distributed Processing Symposium (2000)","DOI":"10.1109\/IPDPS.2000.846052"},{"key":"25_CR33","unstructured":"J\u00fclich Supercomputing Centre: BGAS user documentation. https:\/\/trac.version.fz-juelich.de\/EIC\/wiki\/bgas-user"},{"key":"25_CR34","unstructured":"J\u00fclich Supercomputing Centre: Blue Gene Active Storage boosts I\/O performance at JSC. http:\/\/www.fz-juelich.de\/SharedDocs\/Pressemitteilungen\/UK\/EN\/2013\/13-11-18bgas.html"},{"key":"25_CR35","doi-asserted-by":"crossref","unstructured":"Kulkarni, A., Manzanares, A., Ionkov, L., Lang, M., Lumsdaine, A.: The design and implementation of a multi-level content-addressable checkpoint file system. In: 2012 19th International Conference on High Performance Computing, pp. 1\u201310, December 2012","DOI":"10.1109\/HiPC.2012.6507514"},{"key":"25_CR36","doi-asserted-by":"crossref","unstructured":"Li, D., Vetter, J.S., Marin, G., McCurdy, C., Cira, C., Liu, Z., Yu, W.: Identifying opportunities for byte-addressable non-volatile memory in extreme-scale scientific applications. In: Proceedings of the 2012 IEEE 26th International Parallel and Distributed Processing Symposium, pp. 945\u2013956 (2012)","DOI":"10.1109\/IPDPS.2012.89"},{"key":"25_CR37","doi-asserted-by":"crossref","unstructured":"Liu, N., Cope, J., Carns, P., Carothers, C., Ross, R., Grider, G., Crume, A., Maltzahn, C.: On the role of burst buffers in leadership-class storage systems. In: Proceedings of the 2012 IEEE Conference on Massive Data Storage (2012)","DOI":"10.1109\/MSST.2012.6232369"},{"key":"25_CR38","doi-asserted-by":"crossref","unstructured":"Lu, G., Zheng, Z., Chien, A.A.: When is multi-version checkpointing needed? In: Proceedings of the 3rd Workshop on Fault-Tolerance for HPC at Extreme Scale, pp. 49\u201356 (2013)","DOI":"10.1145\/2465813.2465821"},{"key":"25_CR39","doi-asserted-by":"crossref","unstructured":"Martino, C.D., Kalbarczyk, Z., Iyer, R.K., Baccanico, F., Fullop, J., Kramer, W.: Lessons learned from the analysis of system failures at petascale: the case of Blue Waters. In: Proceedings of the 2014 44th Annual IEEE\/IFIP International Conference on Dependable Systems and Networks, pp. 610\u2013621 (2014)","DOI":"10.1109\/DSN.2014.62"},{"key":"25_CR40","unstructured":"Metzler, B., Trivedi, A.: Prototyping byte-addressable NVM access. In: Proceedings of 11th OpenFabrics Developers Workshop (2015)"},{"key":"25_CR41","doi-asserted-by":"crossref","unstructured":"Moody, A., Bronevetsky, G., Mohror, K., de Supinski, B.R.: Design, modeling, and evaluation of a scalable multi-level checkpointing system. In: Proceedings of the 2010 ACM\/IEEE International Conference for High Performance Computing, Networking, Storage and Analysis, pp. 1\u201311 (2010)","DOI":"10.1109\/SC.2010.18"},{"issue":"2","key":"25_CR42","doi-asserted-by":"crossref","first-page":"203","DOI":"10.1177\/1094342006064503","volume":"20","author":"J Nieplocha","year":"2006","unstructured":"Nieplocha, J., Palmer, B., Tipparaju, V., Krishnan, M., Trease, H., Apr\u00e0, E.: Advances, applications and performance of the global arrays shared memory programming toolkit. Int. J. High Perform. Comput. Appl. 20(2), 203\u2013231 (2006)","journal-title":"Int. J. High Perform. Comput. Appl."},{"key":"25_CR43","doi-asserted-by":"crossref","unstructured":"Numrich, R.W., Reid, J.: Co-Array Fortran for parallel programming. SIGPLAN Fortran Forum 17(2) (1998)","DOI":"10.1145\/289918.289920"},{"key":"25_CR44","doi-asserted-by":"crossref","unstructured":"Ouyang, X., et al.: Enhancing checkpoint performance with staging I\/O and SSD. In: Proceedings of 2010 International Workshop on Storage Network Architecture and Parallel I\/Os, May 2010","DOI":"10.1109\/SNAPI.2010.10"},{"key":"25_CR45","doi-asserted-by":"crossref","first-page":"274","DOI":"10.1016\/j.anucene.2012.06.040","volume":"51","author":"PK Romano","year":"2013","unstructured":"Romano, P.K., Forget, B.: The OpenMC Monte Carlo particle transport code. Ann. Nucl. Energy 51, 274\u2013281 (2013)","journal-title":"Ann. Nucl. Energy"},{"key":"25_CR46","doi-asserted-by":"crossref","unstructured":"Sato, K., Mohror, K., Moody, A., Gamblin, T., de Supinski, B.R., Maruyama, N., Matsuoka, S.: A user-level InfiniBand-based file system and checkpoint strategy for burst buffers. In: Proceedings of 2014 14th IEEE\/ACM International Symposium on Cluster, Cloud and Grid Computing (2014)","DOI":"10.1109\/CCGrid.2014.24"},{"issue":"3","key":"25_CR47","doi-asserted-by":"crossref","first-page":"222","DOI":"10.1145\/357369.357371","volume":"1","author":"RD Schlichting","year":"1983","unstructured":"Schlichting, R.D., Schneider, F.B.: Fail-stop processors: an approach to designing fault-tolerant computing systems. ACM Trans. Comput. Syst. 1(3), 222\u2013238 (1983)","journal-title":"ACM Trans. Comput. Syst."},{"key":"25_CR48","doi-asserted-by":"crossref","unstructured":"Schroeder, B., Gibson, G.A.: A large-scale study of failures in high-performance computing systems. In: Proceedings of 2006 IEEE\/IFIP International Conference on Dependable Systems and Networks (2006)","DOI":"10.1109\/DSN.2006.5"},{"key":"25_CR49","doi-asserted-by":"crossref","unstructured":"Shantharam, M., Srinivasmurthy, S., Raghavan, P.: Characterizing the impact of soft errors on iterative methods in scientific computing. In: Proceedings of Supercomputing (2011)","DOI":"10.1145\/1995896.1995922"},{"key":"25_CR50","doi-asserted-by":"crossref","unstructured":"Young, J.W.: A first order approximation to the optimum checkpoint interval. Commun. ACM 17(9) (1974)","DOI":"10.1145\/361147.361115"},{"key":"25_CR51","doi-asserted-by":"crossref","unstructured":"Zheng, Z., Yu, L., Tang, W., Lan, Z., Gupta, R., Desai, N., Coghlan, S., Buettner, D.: Co-analysis of RAS log and job log on Blue Gene\/P. In: Proceedings of 2011 IEEE International Parallel and Distributed Processing Symposium (2011)","DOI":"10.1109\/IPDPS.2011.83"},{"key":"25_CR52","doi-asserted-by":"crossref","unstructured":"Zhou, M., Du, Y., Childers, B.R., Melhem, R., Mosse, D.: Writeback-aware bandwidth partitioning for multi-core systems with PCM. In: Proceedings of the 22nd International Conference on Parallel Architectures and Compilation Techniques, pp. 113\u2013122 (2013)","DOI":"10.1109\/PACT.2013.6618809"}],"container-title":["Lecture Notes in Computer Science","High Performance Computing"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-41321-1_25","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,9,9]],"date-time":"2019-09-09T14:32:53Z","timestamp":1568039573000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-319-41321-1_25"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2016]]},"ISBN":["9783319413204","9783319413211"],"references-count":52,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-41321-1_25","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2016]]}}}