{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,11]],"date-time":"2025-06-11T22:10:02Z","timestamp":1749679802424,"version":"3.41.0"},"publisher-location":"Cham","reference-count":135,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783319448800"},{"type":"electronic","value":"9783319448817"}],"license":[{"start":{"date-parts":[[2016,1,1]],"date-time":"2016-01-01T00:00:00Z","timestamp":1451606400000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2016]]},"DOI":"10.1007\/978-3-319-44881-7_8","type":"book-chapter","created":{"date-parts":[[2016,10,27]],"date-time":"2016-10-27T07:11:13Z","timestamp":1477552273000},"page":"147-170","source":"Crossref","is-referenced-by-count":0,"title":["High-Performance Storage Support for Scientific Big Data Applications on the Cloud"],"prefix":"10.1007","author":[{"given":"Dongfang","family":"Zhao","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Akash","family":"Mahakode","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sandip","family":"Lakshminarasaiah","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ioan","family":"Raicu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2016,10,28]]},"reference":[{"key":"8_CR1","doi-asserted-by":"crossref","unstructured":"Shvachko, K., Kuang, H., Radia, S., Chansler, R.: The hadoop distributed file system. In: Proceedings of IEEE Symposium on Mass Storage Systems and Technologies (2010)","DOI":"10.1109\/MSST.2010.5496972"},{"key":"8_CR2","doi-asserted-by":"crossref","unstructured":"Carns, P., Lang, S., Ross, R., Vilayannur, M., Kunkel, J., Ludwig, T.: Small-file access in parallel file systems. In: Proceedings of IEEE International Symposium on Parallel and Distributed Processing (2009)","DOI":"10.1109\/IPDPS.2009.5161029"},{"key":"8_CR3","doi-asserted-by":"crossref","unstructured":"Ghemawat, S., Gobioff, H., Leung, S.T.: The Google file system. In: ACM Symposium on Operating Systems Principles (2003)","DOI":"10.1145\/945445.945450"},{"key":"8_CR4","unstructured":"S3FS: https:\/\/code.google.com\/p\/s3fs\/ . Accessed 6 March 2015"},{"key":"8_CR5","unstructured":"FUSE: http:\/\/fuse.sourceforge.net . Accessed 5 Sept 2014"},{"key":"8_CR6","unstructured":"Weil, S.A., Brandt, S.A., Miller, E.L., Long, D.D.E., Maltzahn, C.: Ceph: a scalable, high-performance distributed file system. In: Proceedings of the 7th Symposium on Operating Systems Design and Implementation (2006)"},{"key":"8_CR7","doi-asserted-by":"crossref","unstructured":"Zhao, D., Zhang, Z., Zhou, X., Li, T., Wang, K., Kimpe, D., Carns, P., Ross, R., Raicu, I.: FusionFS: Toward supporting data-intensive scientific applications on extreme-scale distributed systems. In: Proceedings of IEEE International Conference on Big Data, pp. 61\u201370 (2014)","DOI":"10.1109\/BigData.2014.7004214"},{"key":"8_CR8","doi-asserted-by":"publisher","unstructured":"Zhao, D., Liu, N., Kimpe, D., Ross, R., Sun, X.H., Raicu, I.: Towards exploring data-intensive scientific applications at extreme scales through systems and simulations. IEEE Trans. Parallel Distrib. Syst. 1\u201314 (2015). doi: 10.1109\/TPDS.2015.2456896","DOI":"10.1109\/TPDS.2015.2456896"},{"key":"8_CR9","doi-asserted-by":"crossref","unstructured":"Weil, S.A., Brandt, S.A., Miller, E.L., Maltzahn, C.: Crush: controlled, scalable, decentralized placement of replicated data. In: Proceedings of the 2006 ACM\/IEEE Conference on Supercomputing (2006)","DOI":"10.1109\/SC.2006.19"},{"key":"8_CR10","unstructured":"Zhao, D., Raicu, I.: Distributed file systems for exascale computing. In: International Conference for High Performance Computing, Networking, Storage and Analysis (SC \u201912), doctoral showcase (2012)"},{"key":"8_CR11","doi-asserted-by":"crossref","unstructured":"Zhao, D., Burlingame, K., Debains, C., Alvarez-Tabio, P., Raicu, I.: Towards high-performance and cost-effective distributed storage systems with information dispersal algorithms. In: IEEE International Conference on Cluster Computing (2013)","DOI":"10.1109\/CLUSTER.2013.6702655"},{"key":"8_CR12","doi-asserted-by":"crossref","unstructured":"Zhao, D., Shou, C., Malik, T., Raicu, I.: Distributed data provenance for large-scale data-intensive computing. In: IEEE International Conference on Cluster Computing (2013)","DOI":"10.1109\/CLUSTER.2013.6702685"},{"key":"8_CR13","doi-asserted-by":"crossref","unstructured":"Zhao, D., Qiao, K., Raicu, I.: Hycache+: towards scalable high-performance caching middleware for parallel file systems. In: Proceedings of the 14th IEEE\/ACM International Symposium on Cluster, Cloud and Grid Computing, pp. 267\u2013276 (2014)","DOI":"10.1109\/CCGrid.2014.11"},{"key":"8_CR14","doi-asserted-by":"crossref","unstructured":"Zhao, D., Raicu, I.: HyCache: a user-level caching middleware for distributed file systems. In: Proceedings of IEEE 27th International Symposium on Parallel and Distributed Processing Workshops and PhD Forum (2013)","DOI":"10.1109\/IPDPSW.2013.83"},{"key":"8_CR15","doi-asserted-by":"crossref","unstructured":"Zhao, D., Yin, J., Qiao, K., Raicu, I.: Virtual chunks: on supporting random accesses to scientific data in compressible storage systems. In: Proceedings of IEEE International Conference on Big Data, pp. 231\u2013240 (2014)","DOI":"10.1109\/BigData.2014.7004238"},{"key":"8_CR16","unstructured":"Zhao, D., Yin, J., Raicu, I.: Improving the i\/o throughput for data-intensive scientific applications with efficient compression mechanisms. In: International Conference for High Performance Computing, Networking, Storage and Analysis (SC \u201913), poster session (2013)"},{"key":"8_CR17","doi-asserted-by":"crossref","unstructured":"Zhao, D., Qiao, K., Zhou, Z., Li, T., Zhou, X., Wang, K., Raicu, I.: Exploiting multi-cores for efficient interchange of large messages in distributed systems. Concurrency Comput.: Pract. Experience 2015 (accepted)","DOI":"10.1002\/cpe.3742"},{"key":"8_CR18","unstructured":"Kodiak: https:\/\/www.nmc-probe.org\/wiki\/Machines:Kodiak . Accessed 5 Sept 2014"},{"key":"8_CR19","unstructured":"Amazon EC2: http:\/\/aws.amazon.com\/ec2 . Accessed 6 March 2015"},{"key":"8_CR20","doi-asserted-by":"crossref","unstructured":"Welch, B., Noer, G.: Optimizing a hybrid SSD\/HDD HPC storage system based on file size distributions. In: IEEE 29th Symposium on Mass Storage Systems and Technologies (2013)","DOI":"10.1109\/MSST.2013.6558449"},{"key":"8_CR21","doi-asserted-by":"crossref","unstructured":"Nagle, D., Serenyi, D., Matthews, A.: The Panasas activescale storage cluster: delivering scalable high bandwidth storage. In: Proceedings of ACM\/IEEE Conference on Supercomputing (2004)","DOI":"10.1109\/SC.2004.57"},{"key":"8_CR22","unstructured":"Zhao, D., Zhang, D., Wang, K., Raicu, I.: Exploring reliability of exascale systems through simulations. In: Proceedings of the 21st ACM\/SCS High Performance Computing Symposium (HPC) (2013)"},{"key":"8_CR23","unstructured":"Schmuck, F., Haskin, R.: GPFS: a shared-disk file system for large computing clusters. In: Proceedings of the 1st USENIX Conference on File and Storage Technologies (2002)"},{"key":"8_CR24","unstructured":"Schwan, P.: Lustre: building a file system for 1,000-node clusters. In: Proceedings of the Linux Symposium (2003)"},{"key":"8_CR25","doi-asserted-by":"crossref","unstructured":"Wu, H., Ren, S., Garzoglio, G., Timm, S., Bernabeu, G., Chadwick, K., Noh, S.-Y.: A reference model for virtual machine launching overhead. IEEE Trans. Cloud Comput. (pp. 99), 1\u20131 (2014)","DOI":"10.1109\/CCGrid.2014.87"},{"key":"8_CR26","doi-asserted-by":"crossref","unstructured":"Wu, H., Ren, S., Garzoglio, G., Timm, S., Bernabeu, G., Noh, S.-Y.: Modeling the virtual machine launching overhead under fermicloud. In: 14th IEEE\/ACM International Symposium on Cluster, Cloud and Grid Computing (CCGrid), May 2014","DOI":"10.1109\/CCGrid.2014.87"},{"key":"8_CR27","doi-asserted-by":"crossref","unstructured":"Li, T., Zhou, X., Brandstatter, K., Zhao, D., Wang, K., Rajendran, A., Zhang, Z., Raicu, I.: ZHT: A light-weight reliable persistent dynamic scalable zero-hop distributed hash table. In: Proceedings of IEEE International Symposium on Parallel and Distributed Processing (2013)","DOI":"10.1109\/IPDPS.2013.110"},{"key":"8_CR28","doi-asserted-by":"crossref","unstructured":"Li, T., Ma, C., Li, J., Zhou, X., Wang, K., Zhao, D., Raicu, I.: Graph\/z: a key-value store based scalable graph processing system. In: IEEE International Conference on Cluster Computing (2015)","DOI":"10.1109\/CLUSTER.2015.90"},{"key":"8_CR29","doi-asserted-by":"crossref","unstructured":"Zhao, Y., Hategan, M., Clifford, B., Foster, I., von Laszewski, G., Nefedova, V., Raicu, I., Stef-Praun, T., Wilde, M.: Swift: Fast, reliable, loosely coupled parallel computation. In: IEEE Congress on Services (2007)","DOI":"10.1109\/SERVICES.2007.63"},{"key":"8_CR30","doi-asserted-by":"crossref","unstructured":"Raicu, I., Foster, I.T., Zhao, Y., Little, P., Moretti, C.M., Chaudhary, A., Thain, D.: The quest for scalable support of data-intensive workloads in distributed systems. In: Proceedings of ACM International Symposium on High Performance Distributed Computing (2009)","DOI":"10.1145\/1551609.1551642"},{"key":"8_CR31","unstructured":"Shou, C., Zhao, D., Malik, T., Raicu, I.: Towards a provenance-aware distributed filesystem. In: 5th Workshop on the Theory and Practice of Provenance (TaPP) (2013)"},{"key":"8_CR32","unstructured":"Protocol Buffers: http:\/\/code.google.com\/p\/protobuf\/ . Accessed 5 Sept 2014"},{"key":"8_CR33","unstructured":"Carns, P.H., Ligon, W.B., Ross, R.B., Thakur, R.: PVFS: a parallel file system for linux clusters. In: Proceedings of the 4th Annual Linux Showcase and Conference (2000)"},{"key":"8_CR34","doi-asserted-by":"crossref","unstructured":"Li, T., Zhou, X., Wang, K., Zhao, D., Sadooghi, I., Zhang, Z., Raicu, I.: A convergence of key-value storage systems from clouds to supercomputer. Concurrency Comput.: Pract. Experience (2016)","DOI":"10.1002\/cpe.3614"},{"key":"8_CR35","doi-asserted-by":"crossref","unstructured":"Zhao, D., Yang, X., Sadooghi, I., Garzoglio, G., Timm, S., Raicu, I.: High-performance storage support for scientific applications on the cloud. In: Proceedings of the 6th Workshop on Scientific Cloud Computing (ScienceCloud) (2015)","DOI":"10.1145\/2755644.2755648"},{"key":"8_CR36","doi-asserted-by":"crossref","unstructured":"Li, T., Keahey, K., Wang, K., Zhao, D., Raicu, I.: A dynamically scalable cloud data infrastructure for sensor networks. In: Proceedings of the 6th Workshop on Scientific Cloud Computing (ScienceCloud) (2015)","DOI":"10.1145\/2755644.2755650"},{"key":"8_CR37","doi-asserted-by":"crossref","unstructured":"Raicu, I., Zhao, Y., Foster, I.T., Szalay, A.: Accelerating large-scale data exploration through data diffusion. In: Proceedings of the 2008 International Workshop on Data-aware Distributed Computing (2008)","DOI":"10.1145\/1383519.1383521"},{"key":"8_CR38","doi-asserted-by":"crossref","unstructured":"Li, S., Huang, H.H.: Black-box performance modeling for solid-state drives. In: 2010 IEEE International Symposium on Modeling, Analysis Simulation of Computer and Telecommunication Systems (MASCOTS) (2010)","DOI":"10.1109\/MASCOTS.2010.48"},{"key":"8_CR39","doi-asserted-by":"crossref","unstructured":"Rizvi, S., Chung, T.-S.: Flash SSD vs HDD: High performance oriented modern embedded and multimedia storage systems. In: 2nd International Conference on Computer Engineering and Technology (ICCET) (2010)","DOI":"10.1109\/ICCET.2010.5485421"},{"key":"8_CR40","doi-asserted-by":"crossref","unstructured":"Chen, F., Koufaty, D.A., Zhang, X.: Hystor: making the best use of solid state drives in high performance storage systems. In: Proceedings of the International Conference on Supercomputing (2011)","DOI":"10.1145\/1995896.1995902"},{"key":"8_CR41","unstructured":"Guerra, J., Pucha, H., Glider, J., Belluomini, W., Rangaswami, R.: Cost effective storage using extent based dynamic tiering. In: Proceedings of the 9th USENIX Conference on File and Stroage Technologies (2011)"},{"key":"8_CR42","doi-asserted-by":"crossref","unstructured":"Zhang, X., Davis, K., Jiang, S.: iTransformer: using SSD to improve disk scheduling for high-performance I\/O. In: Proceedings of the 2012 IEEE 26th International Parallel and Distributed Processing Symposium (2012)","DOI":"10.1109\/IPDPS.2012.70"},{"key":"8_CR43","doi-asserted-by":"crossref","unstructured":"Zhang, X., Ke, L., Davis, K., Jiang, S.: iBridge: improving unaligned parallel file access with solid-state drives. In: Proceedings of the 2013 IEEE 27th International Parallel and Distributed Processing Symposium (2013)","DOI":"10.1109\/IPDPS.2013.21"},{"key":"8_CR44","doi-asserted-by":"crossref","unstructured":"Mao, B., Jiang, H., Feng, D., Wu, S., Chen, J., Zeng, L., Tian, L.: HPDA: a hybrid parity-based disk array for enhanced performance and reliability. In: 2010 IEEE International Symposium on Parallel Distributed Processing (IPDPS) (2010)","DOI":"10.1109\/IPDPS.2010.5470361"},{"key":"8_CR45","unstructured":"Badam, A., Pai, V.S.: SSDAlloc: hybrid SSD\/RAM memory management made easy. In: Proceedings of the 8th USENIX Conference on Networked systems design and implementation (2011)"},{"key":"8_CR46","doi-asserted-by":"crossref","unstructured":"Wang, C., Vazhkudai, S.S., Ma, X., Meng, F., Kim, Y., Engelmann, C.: Nvmalloc: exposing an aggregate ssd store as a memory partition in extreme-scale machines. In: Proceedings of the 2012 IEEE 26th International Parallel and Distributed Processing Symposium (2012)","DOI":"10.1109\/IPDPS.2012.90"},{"key":"8_CR47","doi-asserted-by":"crossref","unstructured":"Wu, X., Narasimha Reddy, A.L.: SCMFS: a file system for storage class memory. In: Proceedings of International Conference for High Performance Computing, Networking, Storage and Analysis (2011)","DOI":"10.1145\/2063384.2063436"},{"key":"8_CR48","unstructured":"Joo, Y., Ryu, J., Park, S., Shin, K.G.: FAST: quick application launch on solid-state drives. In: Proceedings of the 9th USENIX Conference on File and Stroage Technologies (2011)"},{"key":"8_CR49","doi-asserted-by":"crossref","unstructured":"Yang, Q., Ren, J.: I-CASH: intelligently coupled array of SSD and HDD. In: Proceedings of the 2011 IEEE 17th International Symposium on High Performance Computer Architecture (2011)","DOI":"10.1109\/HPCA.2011.5749736"},{"key":"8_CR50","doi-asserted-by":"crossref","unstructured":"Fares, R., Romoser, B., Zong, Z., Nijim, M., Qin, X.: Performance evaluation of traditional caching policies on a large system with petabytes of data. In: 2012 IEEE 7th International Conference on Networking, Architecture and Storage (NAS) (2012)","DOI":"10.1109\/NAS.2012.32"},{"key":"8_CR51","doi-asserted-by":"crossref","unstructured":"Podlipnig, S., B\u00f6sz\u00f6rmenyi, L.: A survey of web cache replacement strategies. ACM Comput. Surv. 35(4) (2003)","DOI":"10.1145\/954339.954341"},{"key":"8_CR52","doi-asserted-by":"crossref","unstructured":"Shi, L., Liu, Z., Xu, L.: Bwcc: a fs-cache based cooperative caching system for network storage system. In: Proceedings of the 2012 IEEE International Conference on Cluster Computing (2012)","DOI":"10.1109\/CLUSTER.2012.41"},{"key":"8_CR53","unstructured":"Wu, C., Xubin, H., Qiang, C., Changsheng, X., Shenggang, W.: Hint-k: an efficient multi-level cache using k-step hints. IEEE Trans. Parallel Distrib. Syst. 99 (2013)"},{"key":"8_CR54","doi-asserted-by":"crossref","unstructured":"Meister, D., Kaiser, J., Brinkmann, A.: Block locality caching for data deduplication. In: Proceedings of the 6th International Systems and Storage Conference (2013)","DOI":"10.1145\/2485732.2485748"},{"key":"8_CR55","doi-asserted-by":"crossref","unstructured":"Xia, P., Feng, D., Jiang, H., Tian, L., Wang, F.: Farmer: a novel approach to file access correlation mining and evaluation reference model for optimizing peta-scale file system performance. In: Proceedings of the 17th International Symposium on High Performance Distributed Computing (2008)","DOI":"10.1145\/1383422.1383445"},{"key":"8_CR56","doi-asserted-by":"crossref","unstructured":"Lin, J., Lu, Q., Ding, X., Zhang, Z., Zhang, X., Sadayappan, P.: Enabling software management for multicore caches with a lightweight hardware support. In: Proceedings of the 2009 ACM\/IEEE Conference on Supercomputing (2009)","DOI":"10.1145\/1654059.1654074"},{"key":"8_CR57","doi-asserted-by":"crossref","unstructured":"Zhan, D., Jiang, H., Seth, S.C.: Locality & utility co-optimization for practical capacity management of shared last level caches. In: Proceedings of the 26th ACM International Conference on Supercomputing (2012)","DOI":"10.1145\/2304576.2304615"},{"key":"8_CR58","doi-asserted-by":"crossref","unstructured":"Gonzalez-Ferez, P., Piernas, J., Cortes, T.: The ram enhanced disk cache project (redcap). In: Proceedings of the 24th IEEE Conference on Mass Storage Systems and Technologies (2007)","DOI":"10.1109\/MSST.2007.4367981"},{"key":"8_CR59","doi-asserted-by":"crossref","unstructured":"Huang, S., Wei, Q., Chen, J., Chen, C., Feng, D.: Improving flash-based disk cache with lazy adaptive replacement. In: 2013 IEEE 29th Symposium on Mass Storage Systems and Technologies (MSST) (2013)","DOI":"10.1109\/MSST.2013.6558447"},{"key":"8_CR60","doi-asserted-by":"crossref","unstructured":"Zhu, Z., Zhang, X.: Access-mode predictions for low-power cache design. IEEE Micro 22(2) (2002)","DOI":"10.1109\/MM.2002.997880"},{"key":"8_CR61","doi-asserted-by":"crossref","unstructured":"Yue, J., Zhu, Y., Cai, Z., Lin, L.: Energy and thermal aware buffer cache replacement algorithm. In: Proceedings of the 2010 IEEE 26th Symposium on Mass Storage Systems and Technologies (MSST) (2010)","DOI":"10.1109\/MSST.2010.5496982"},{"key":"8_CR62","doi-asserted-by":"crossref","unstructured":"Manzanares, A., Ruan, X., Yin, S., Xie, J., Ding, Z., Tian, Y., Majors, J., Qin, X.: Energy efficient prefetching with buffer disks for cluster file systems. In: Proceedings of the 2010 39th International Conference on Parallel Processing (2010)","DOI":"10.1109\/ICPP.2010.48"},{"key":"8_CR63","doi-asserted-by":"crossref","unstructured":"Li, Z., Wilson, C., Jiang, Z., Liu, Y., Zhao, B., Jin, C., Zhang, Z.L., Dai, Y.: Efficient batched synchronization in dropbox-like cloud storage services. In: Proceedings of the 14th International Middleware Conference (2013)","DOI":"10.1007\/978-3-642-45065-5_16"},{"key":"8_CR64","doi-asserted-by":"crossref","unstructured":"Xu, Y., Xing, C., Zhou, L.: A cache replacement algorithm in hierarchical storage of continuous media object. In: Advances in Web-Age Information Management: 5th International Conference (2004)","DOI":"10.1007\/978-3-540-27772-9_17"},{"key":"8_CR65","doi-asserted-by":"crossref","unstructured":"Li, R., Guo, R., Xu, Z., Feng, W.: A prefetching model based on access popularity for geospatial data in a cluster-based caching system. Int. J. Geogr. Inf. Sci. 26(10) (2012)","DOI":"10.1080\/13658816.2012.659184"},{"key":"8_CR66","doi-asserted-by":"crossref","unstructured":"Qiao, K., Tao, F., Zhang, L., Li, Z.: A ga maintained by binary heap and transitive reduction for addressing psp. In: 2010 International Conference on Intelligent Computing and Integrated Systems (ICISS) (2010)","DOI":"10.1109\/ICISS.2010.5654994"},{"key":"8_CR67","doi-asserted-by":"crossref","unstructured":"Tao, F., Qiao, K., Zhang, L., Li, Z., Nee, A.: GA-BHTR: an improved genetic algorithm for partner selection in virtual manufacturing. Int. J. Prod. Res. 50(8) (2012)","DOI":"10.1080\/00207543.2011.561883"},{"key":"8_CR68","doi-asserted-by":"crossref","unstructured":"Calinescu, G., Qiao, K.: Asymmetric topology control: exact solutions and fast approximations. In: IEEE International Conference on Computer Communications (INFOCOM \u201912) (2012)","DOI":"10.1109\/INFCOM.2012.6195825"},{"key":"8_CR69","doi-asserted-by":"crossref","unstructured":"Calinescu, G., Kapoor, S., Qiao, K., Shin, J.: Stochastic strategic routing reduces attack effects. In: Global Telecommunications Conference (GLOBECOM 2011), 2011. IEEE (2011)","DOI":"10.1109\/GLOCOM.2011.6133863"},{"key":"8_CR70","doi-asserted-by":"crossref","unstructured":"Zhao, D,, Yang, L.: Incremental isometric embedding of high-dimensional data using connected neighborhood graphs. IEEE Trans. Pattern Anal. Mach. Intell. 31(1), 86\u201398 (2009)","DOI":"10.1109\/TPAMI.2008.34"},{"key":"8_CR71","doi-asserted-by":"crossref","unstructured":"Lohfert, R., Lu, J., Zhao, D.; Solving sql constraints by incremental translation to sat. In: International Conference on Industrial, Engineering and Other Applications of Applied Intelligent Systems (2008)","DOI":"10.1007\/978-3-540-69052-8_70"},{"key":"8_CR72","doi-asserted-by":"crossref","unstructured":"Zhao, D., Yang, L.: Incremental construction of neighborhood graphs for nonlinear dimensionality reduction. In: Proceedings of 18th International Conference on Pattern Recognition, vol.\u00a03, pp. 177\u2013180 (2006)","DOI":"10.1109\/ICPR.2006.707"},{"key":"8_CR73","doi-asserted-by":"crossref","unstructured":"Ferreira, K.B., Riesen, R., Arnold, D., Ibtesham, D., Brightwell, R.: The viability of using compression to decrease message log sizes. In: Proceedings of International Conference on Parallel Processing Workshops (2013)","DOI":"10.1007\/978-3-642-36949-0_54"},{"key":"8_CR74","doi-asserted-by":"crossref","unstructured":"Zerin Islam, T., Mohror, K., Bagchi, S., Moody, A., de\u00a0Supinski, B.R., Eigenmann, R.: McrEngine: a scalable checkpointing system using data-aware aggregation and compression. In: Proceedings of the International Conference on High Performance Computing, Networking, Storage and Analysis (SC) (2012)","DOI":"10.1109\/SC.2012.77"},{"key":"8_CR75","doi-asserted-by":"crossref","unstructured":"Slim Bouguerra, M., Gainaru, A., Gomez, L.B., Cappello, F., Matsuoka, S., Maruyam, N.: Improving the computing efficiency of hpc systems using a combination of proactive and preventive checkpointing. In: IEEE International Symposium on Parallel Distributed Processing (2013)","DOI":"10.1109\/IPDPS.2013.74"},{"key":"8_CR76","doi-asserted-by":"crossref","unstructured":"Noeth, M., Marathe, J., Mueller, F., Schulz, M., de\u00a0Supinski, B.: Scalable compression and replay of communication traces in massively parallel environments. In: Proceedings of the 2006 ACM\/IEEE Conference on Supercomputing (SC) (2006)","DOI":"10.1145\/1188455.1188605"},{"key":"8_CR77","doi-asserted-by":"crossref","unstructured":"Laney, D., Langer, S., Weber, C., Lindstrom, P., Wegener, A.: Assessing the effects of data compression in simulations using physically motivated metrics. In: Proceedings of the International Conference on High Performance Computing, Networking, Storage and Analysis (2013)","DOI":"10.1145\/2503210.2503283"},{"key":"8_CR78","doi-asserted-by":"crossref","unstructured":"Lakshminarasimhan, S., Jenkins, J., Arkatkar, I., Gong, Z., Kolla, H., Ku, S.-H., Ethier, S., Chen, J., Chang, C.S., Klasky, S., Latham, R., Ross, R., Samatova, N.F.: ISABELA-QA: query-driven analytics with ISABELA-compressed extreme-scale scientific data. In: Proceedings of 2011 International Conference for High Performance Computing, Networking, Storage and Analysis (SC\u201911) (2011)","DOI":"10.1145\/2063384.2063425"},{"key":"8_CR79","unstructured":"MPEG-1: http:\/\/en.wikipedia.org\/wiki\/MPEG-1 . Accessed 5 Sept 2014"},{"key":"8_CR80","doi-asserted-by":"crossref","unstructured":"Bicer, T., Yin, J., Chiu, D., Agrawal, G., Schuchardt, K.: Integrating online compression to accelerate large-scale data analytics applications. In: Proceedings of the 2013 IEEE 27th International Symposium on Parallel and Distributed Processing (IPDPS) (2013)","DOI":"10.1109\/IPDPS.2013.81"},{"key":"8_CR81","doi-asserted-by":"crossref","unstructured":"Schendel, E.R., Pendse, S.V., Jenkins, J., Boyuka, D.A., II, Gong, Z., Lakshminarasimhan, S., Liu, Q., Kolla, H., Chen, J., Klasky, S.,Ross, R., Samatova, N.F.: Isobar hybrid compression-i\/o interleaving for large-scale parallel i\/o optimization, In: Proceedings of International Symposium on High-Performance Parallel and Distributed Computing (2012)","DOI":"10.1145\/2287076.2287086"},{"key":"8_CR82","doi-asserted-by":"crossref","unstructured":"Jenkins, J., Schendel, E.R., Lakshminarasimhan, S., Boyuka, D.S., II, Rogers, T., Ethier, S., Ross, R., Klasky, S., Samatova, N.F.: Byte-precision level of detail processing for variable precision analytics. In: Proceedings of the International Conference on High Performance Computing, Networking, Storage and Analysis (SC) (2012)","DOI":"10.1109\/SC.2012.26"},{"key":"8_CR83","doi-asserted-by":"crossref","unstructured":"Burrows, M., Jerian, C., Lampson, B., Mann, T.: On-line data compression in a log-structured file system. In: Proceedings of the Fifth International Conference on Architectural Support for Programming Languages and Operating Systems (ASPLOS) (1992)","DOI":"10.1145\/143365.143376"},{"key":"8_CR84","volume-title":"MacDonald","author":"P Joshua","year":"2000","unstructured":"Joshua, P.: MacDonald. File system support for delta compression. Technical report, University of California, Berkley (2000)"},{"key":"8_CR85","unstructured":"Olson, M.A., Bostic, K., Seltzer M.: db. In: Proceedings of the Annual Conference on USENIX Annual Technical Conference (1999)"},{"key":"8_CR86","doi-asserted-by":"crossref","unstructured":"Edel, N.K., Tuteja, D., Miller, E.L., Brandt S.A.: Mramfs: a compressing file system for non-volatile ram. In: Proceedings of the the IEEE Computer Society\u2019s 12th Annual International Symposium on Modeling, Analysis, and Simulation of Computer and Telecommunications Systems (MASCOTS) (2004)","DOI":"10.1109\/MASCOT.2004.1348317"},{"key":"8_CR87","doi-asserted-by":"crossref","unstructured":"Muthitacharoen, A., Chen, B., Mazi\u00e8res, D.: A low-bandwidth network file system. In: Proceedings of the Eighteenth ACM Symposium on Operating Systems Principles (SOSP) (2001)","DOI":"10.1145\/502034.502052"},{"key":"8_CR88","unstructured":"Park, K.S., Ihm, S., Bowman, M., Pai, V.S.: Supporting practical content-addressable caching with czip compression. In: 2007 USENIX Annual Technical Conference (2007)"},{"key":"8_CR89","unstructured":"Meister, D., Brinkmann, A., S\u00fc\u00df, T.: File recipe compression in data deduplication systems. In: Proceedings of the 11th USENIX Conference on File and Storage Technologies (FAST) (2013)"},{"key":"8_CR90","doi-asserted-by":"crossref","unstructured":"Lakshminarasimhan, S., Boyuka, D.A., Pendse, S.V., Zou, X., Jenkins, J., Vishwanath, V., Papka, M.E., Samatova, N.F.: Scalable in situ scientific data encoding for analytical query processing. In: Proceedings of the 22nd International Symposium on High-performance Parallel and Distributed Computing (HPDC) (2013)","DOI":"10.1145\/2493123.2465527"},{"key":"8_CR91","doi-asserted-by":"crossref","unstructured":"Gong, Z., Lakshminarasimhan, S., Jenkins, J., Kolla, H., Ethier, S., Chen, J., Ross, R., Klasky, S., Samatova, N.F.: Multi-level layout optimization for efficient spatio-temporal queries on isabela-compressed data. In: Proceedings of the 2012 IEEE 26th International Parallel and Distributed Processing Symposium (IPDPS) (2012)","DOI":"10.1109\/IPDPS.2012.83"},{"key":"8_CR92","unstructured":"Shnaiderman, L., Shmueli, O.: A parallel twig join algorithm for XML processing using a GPGPU. In: International Workshop on Accelerating Data Management Systems Using Modern Processor and Storage Architectures (2012)"},{"key":"8_CR93","doi-asserted-by":"crossref","unstructured":"Wang, H., Potluri, S., Bureddy, D., Rosales, C., Panda, D.K.: Gpu-aware mpi on rdma-enabled clusters: design, implementation and evaluation. IEEE Trans. Parallel Distrib. Syst. 25(10) (2014)","DOI":"10.1109\/TPDS.2013.222"},{"key":"8_CR94","doi-asserted-by":"crossref","unstructured":"Bordawekar, R., Bondhugula, U., Rao. R.: Believe it or not!: mult-core cpus can match gpu performance for a flop-intensive application! In: Proceedings of the 19th International Conference on Parallel Architectures and Compilation Techniques, PACT \u201910, (2010)","DOI":"10.1145\/1854273.1854340"},{"key":"8_CR95","doi-asserted-by":"crossref","unstructured":"Farooqui, N., Schwan, K., Yalamanchili, S.: Efficient instrumentation of gpgpu applications using information flow analysis and symbolic execution. In: Proceedings of Workshop on General Purpose Processing Using GPUs, GPGPU-7 (2014)","DOI":"10.1145\/2588768.2576782"},{"key":"8_CR96","unstructured":"Muniswamy-Reddy, K.-K.: Foundations for provenance-aware systems (2010)"},{"key":"8_CR97","doi-asserted-by":"crossref","unstructured":"Foster, I.T., Vckler, J.S., Wilde, M., Zhao, Y.: The virtual data grid: a new model and architecture for data-intensive collaboration. In: CIDR\u201903 (2003)","DOI":"10.1109\/SSDM.2003.1214945"},{"key":"8_CR98","unstructured":"Provenance aware service oriented architecture. http:\/\/twiki.pasoa.ecs.soton.ac.uk\/bin\/view\/PASOA\/WebHome . Accessed 6 July 2015"},{"key":"8_CR99","unstructured":"Parker-Wood, A., Long, D.D.E., Miller, E.L., Seltzer, M., Tunkelang, D.: Making sense of file systems through provenance and rich metadata. Technical Report UCSC-SSRC-12-01, University of California, Santa Cruz, March 2012"},{"key":"8_CR100","unstructured":"Muniswamy-Reddy, K.-K., Holland, D.A., Braun, U., Seltzer, M.: Provenance-aware storage systems. In: Proceedings of the annual conference on USENIX \u201906 Annual Technical Conference (2006)"},{"key":"8_CR101","unstructured":"Muniswamy-Reddy, K.-K., Macko, P., Seltzer, M.: Making a cloud provenance-aware. In: 1st Workshop on the Theory and Practice of Provenance (2009)"},{"key":"8_CR102","unstructured":"Muniswamy-Reddy, K.-K., Braun, U., Holland, D.A., Macko, P., Maclean, D., Margo, D., Seltzer, M., Smogor, R.: Layering in provenance systems. In: Proceedings of the 2009 USENIX Annual Technical Conference (2009)"},{"key":"8_CR103","doi-asserted-by":"crossref","unstructured":"Gehani, A., Tariq, D.: SPADE: support for provenance auditing in distributed environments. In: Proceedings of the 13th International Middleware Conference (2012)","DOI":"10.1007\/978-3-642-35170-9_6"},{"key":"8_CR104","doi-asserted-by":"crossref","unstructured":"Zhou, W., Sherr, M., Tao, T., Li, X., Thau Loo, B., Mao, Y.: Efficient querying and maintenance of network provenance at internet-scale. In: Proceedings of the 2010 International Conference on Management of Data, pp. 615\u2013626 (2010)","DOI":"10.1145\/1807167.1807234"},{"key":"8_CR105","doi-asserted-by":"crossref","unstructured":"Abraham, J., Brazier, P., Chebotko, A., Navarro, J., Piazza, A.: Distributed storage and querying techniques for a semantic web of scientific workflow provenance. In: 2010 IEEE International Conference on Services Computing (SCC), pp. 178\u2013185. IEEE (2010)","DOI":"10.1109\/SCC.2010.14"},{"key":"8_CR106","doi-asserted-by":"crossref","unstructured":"Malik, T., Gehani, A., Tariq, D., Zaffar, F.: Sketching distributed data provenance. In: Data Provenance and Data Management in eScience, pp. 85\u2013107 (2013)","DOI":"10.1007\/978-3-642-29931-5_4"},{"key":"8_CR107","doi-asserted-by":"crossref","unstructured":"Heinis, T., Alonso, G.: Efficient lineage tracking for scientific workflows. In: Proceedings of the 2008 ACM SIGMOD International Conference on Management of Data, pp. 1007\u20131018 (2008)","DOI":"10.1145\/1376616.1376716"},{"key":"8_CR108","unstructured":"Extensible Markup Language (XML): http:\/\/www.w3.org\/xml\/ . Accessed 13 Dec 2014"},{"key":"8_CR109","unstructured":"JSON: http:\/\/www.json.org\/ . Accessed 8 Dec 2014"},{"key":"8_CR110","unstructured":"Binary JSON: http:\/\/bsonspec.org\/ . Accessed 13 Dec 2014"},{"key":"8_CR111","unstructured":"Apache Thrift: https:\/\/thrift.apache.org\/ . Accessed 8 Dec 2014"},{"key":"8_CR112","unstructured":"Apache Avro: http:\/\/avro.apache.org\/ . Accessed 13 Dec 2014"},{"key":"8_CR113","unstructured":"Apache Etch: https:\/\/etch.apache.org\/ . Accessed 13 Dec 2014"},{"key":"8_CR114","unstructured":"BERT: http:\/\/bert-rpc.org\/ . Accessed 13 Dec 2014"},{"key":"8_CR115","unstructured":"Message Pack: http:\/\/msgpack.org\/ . Accessed 13 Dec 2014"},{"key":"8_CR116","unstructured":"Hessian: http:\/\/hessian.caucho.com\/ . Accessed 13 Dec 2014"},{"key":"8_CR117","unstructured":"ICE: http:\/\/doc.zeroc.com\/display\/ice34\/home . Accessed 13 Dec 2014"},{"key":"8_CR118","unstructured":"CBOR: http:\/\/cbor.io\/ . Accessed 13 Dec 2014"},{"key":"8_CR119","unstructured":"Dean, J., Ghemawat, S.: MapReduce: simplified data processing on large clusters. In: Proceedings of USENIX Symposium on Opearting Systems Design & Implementation (2004)"},{"key":"8_CR120","unstructured":"Apache Hadoop: http:\/\/hadoop.apache.org\/ . Accessed 5 Sept 2014"},{"key":"8_CR121","unstructured":"Zaharia, M., Chowdhury, M., Franklin, M.J., Shenker, S., Stoica, I.: Spark: cluster computing with working sets. In: Proceedings of the 2nd USENIX Conference on Hot Topics in Cloud Computing (2010)"},{"key":"8_CR122","unstructured":"MPICH: http:\/\/www.mpich.org\/ . Accessed 10 Dec 2014"},{"key":"8_CR123","unstructured":"Open MPI: http:\/\/www.open-mpi.org\/ . Accessed 10 Dec 2014"},{"key":"8_CR124","unstructured":"OpenMP: http:\/\/openmp.org\/wp\/ . Accessed 9 Dec 2014"},{"key":"8_CR125","unstructured":"PPL: http:\/\/msdn.microsoft.com\/en-us\/library\/dd492418.aspx . Accessed 13 Dec 2014"},{"key":"8_CR126","doi-asserted-by":"crossref","unstructured":"Jeon, M., He, Y., Elnikety, S., Cox, A.L., Rixner, S.: Adaptive parallelism for web search. In: Proceedings of the 8th ACM European Conference on Computer Systems, EuroSys \u201913 (2013)","DOI":"10.1145\/2465351.2465367"},{"key":"8_CR127","doi-asserted-by":"crossref","unstructured":"Jeon, M., Kim, S., Hwang, S., He, Y., Elnikety, S., Cox, A.L., Rixner, S.: Predictive parallelization: taming tail latencies in web search. In: Proceedings of the 37th International ACM SIGIR Conference on Research & Development in Information Retrieval, SIGIR \u201914 (2014)","DOI":"10.1145\/2600428.2609572"},{"key":"8_CR128","unstructured":"Lee, J., Winslett, M., Ma, X., Yu, S.: Enhancing data migration performance via parallel data compression. In: Proceedings of the 16th International Parallel and Distributed Processing Symposium, IPDPS \u201902 (2002)"},{"key":"8_CR129","doi-asserted-by":"crossref","unstructured":"Klasky, S., Ethier, S., Lin, Z., Martins, K., McCune, D., Samtaney, R.: Grid-based parallel data streaming implemented for the gyrokinetic toroidal code. In: Proceedings of the 2003 ACM\/IEEE Conference on Supercomputing, SC \u201903 (2003)","DOI":"10.1145\/1048935.1050175"},{"key":"8_CR130","doi-asserted-by":"crossref","unstructured":"Warneke, D., Kao, O.: Nephele: efficient parallel data processing in the cloud. In: Proceedings of the 2nd Workshop on Many-Task Computing on Grids and Supercomputers, MTAGS \u201909 (2009)","DOI":"10.1145\/1646468.1646476"},{"key":"8_CR131","unstructured":"Yu, Y., Isard, M., Fetterly, D., Budiu, M., Erlingsson, U., Gunda, P.K., Currey, J.: Dryadlinq: a system for general-purpose distributed data-parallel computing using a high-level language. In: Proceedings of the 8th USENIX Conference on Operating Systems Design and Implementation, OSDI\u201908 (2008)"},{"key":"8_CR132","doi-asserted-by":"crossref","unstructured":"Ronnie, C., Bob, J., Per-Ake, L., Bill, R., Darren, S., Simon, W., Jingren, Z.: Scope: easy and efficient parallel processing of massive data sets. Proc. VLDB Endow. 1(2), 1265\u20131276 (2008)","DOI":"10.14778\/1454159.1454166"},{"key":"8_CR133","doi-asserted-by":"crossref","unstructured":"Ahrens, J., Brislawn, K., Martin, K., Geveci, B., Charles Law, C., Papka, M.: Large-scale data visualization using parallel data streaming. In: Computer Graphics and Applications. IEEE, 21(4), July 2001","DOI":"10.1109\/38.933522"},{"key":"8_CR134","doi-asserted-by":"crossref","unstructured":"Allen, M.D., Sridharan, S., Sohi, G.S.: Serialization sets: a dynamic dependence-based parallel execution model. In: Proceedings of the 14th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming, PPoPP \u201909 (2009)","DOI":"10.1145\/1594835.1504190"},{"key":"8_CR135","doi-asserted-by":"crossref","unstructured":"Voss, M., Eigenmann, R.: Reducing parallel overheads through dynamic serialization. In: Proceedings of the 13th International Symposium on Parallel Processing and the 10th Symposium on Parallel and Distributed Processing, IPPS \u201999\/SPDP \u201999 (1999)","DOI":"10.1109\/IPPS.1999.760440"}],"container-title":["Computer Communications and Networks","Resource Management for Big Data Platforms"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-44881-7_8","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,11]],"date-time":"2025-06-11T21:41:58Z","timestamp":1749678118000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-319-44881-7_8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2016]]},"ISBN":["9783319448800","9783319448817"],"references-count":135,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-44881-7_8","relation":{},"ISSN":["1617-7975","2197-8433"],"issn-type":[{"type":"print","value":"1617-7975"},{"type":"electronic","value":"2197-8433"}],"subject":[],"published":{"date-parts":[[2016]]}}}