{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,27]],"date-time":"2026-06-27T00:17:10Z","timestamp":1782519430108,"version":"3.54.5"},"reference-count":25,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2022,12,26]],"date-time":"2022-12-26T00:00:00Z","timestamp":1672012800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,12,26]],"date-time":"2022-12-26T00:00:00Z","timestamp":1672012800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Sci. China Inf. Sci."],"published-print":{"date-parts":[[2023,1]]},"DOI":"10.1007\/s11432-021-3406-5","type":"journal-article","created":{"date-parts":[[2023,1,4]],"date-time":"2023-01-04T06:03:09Z","timestamp":1672812189000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":7,"title":["LPW: an efficient data-aware cache replacement strategy for Apache Spark"],"prefix":"10.1007","volume":"66","author":[{"given":"Hui","family":"Li","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shuping","family":"Ji","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hua","family":"Zhong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wei","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lijie","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhen","family":"Tang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jun","family":"Wei","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tao","family":"Huang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2022,12,26]]},"reference":[{"key":"3406_CR1","doi-asserted-by":"crossref","unstructured":"Shvachko K, Kuang H, Radia S, et al. The Hadoop distributed file system. In: Proceedings of Symposium on Mass Storage Systems and Technologies, 2010. 1\u201310","DOI":"10.1109\/MSST.2010.5496972"},{"key":"3406_CR2","doi-asserted-by":"crossref","unstructured":"Li H Y, Ghodsi A, Zaharia M, et al. Tachyon: reliable, memory speed storage for cluster computing frameworks. In: Proceedings of the ACM Symposium on Cloud Computing, Seattle, 2014. 1\u201315","DOI":"10.1145\/2670979.2670985"},{"key":"3406_CR3","doi-asserted-by":"crossref","unstructured":"Saha B, Shah H, Seth S, et al. Apache Tez: a unifying framework for modeling and building data processing applications. In: Proceedings of International Conference on Management of Data, Melbourne, 2015. 1357\u20131369","DOI":"10.1145\/2723372.2742790"},{"key":"3406_CR4","first-page":"45","volume":"37","author":"M Zaharia","year":"2012","unstructured":"Zaharia M, Chowdhury M, Das T, et al. Fast and interactive analytics over Hadoop data with spark. Adv Comput Syst Assoc, 2012, 37: 45\u201351","journal-title":"Adv Comput Syst Assoc"},{"key":"3406_CR5","unstructured":"Zaharia M, Chowdhury M, Das T, et al. Resilient distributed datasets: a fault-tolerant abstraction for in-memory cluster computing. In: Proceedings of USENIX Conference on Networked Systems Design and Implementation, 2012"},{"key":"3406_CR6","doi-asserted-by":"crossref","unstructured":"Li H, Wang D, Huang T Z, et al. Detecting cache-related bugs in Spark applications. In: Proceedings of International Symposium on Software Testing and Analysis, Virtual Event, 2020. 363\u2013375","DOI":"10.1145\/3395363.3397353"},{"key":"3406_CR7","doi-asserted-by":"publisher","first-page":"1285","DOI":"10.1007\/s10766-016-0470-1","volume":"45","author":"Y Z Geng","year":"2017","unstructured":"Geng Y Z, Shi X H, Pei C, et al. LCS: an efficient data eviction strategy for Spark. Int J Parallel Prog, 2017, 45: 1285\u20131297","journal-title":"Int J Parallel Prog"},{"key":"3406_CR8","doi-asserted-by":"crossref","unstructured":"Yu Y H, Wang W, Zhang J, et al. LRC: dependency-aware cache management for data analytics clusters. In: Proceedings of Conference on Computer Communications, Atlanta, 2017. 1\u20139","DOI":"10.1109\/INFOCOM.2017.8057007"},{"key":"3406_CR9","doi-asserted-by":"crossref","unstructured":"Huang S S, Huang J, Dai J Q, et al. The Hibench benchmark suite: characterization of the MapReduce-based data analysis. In: Proceedings of International Conference on Data Engineering Workshop, Long Beach, 2010. 41\u201351","DOI":"10.1109\/ICDEW.2010.5452747"},{"key":"3406_CR10","doi-asserted-by":"crossref","unstructured":"Li C, Cox A L. GD-Wheel: a cost-aware replacement policy for key-value stores. In: Proceedings of the European Conference on Computer Systems, Bordeaux, 2015. 1\u201315","DOI":"10.1145\/2741948.2741956"},{"key":"3406_CR11","unstructured":"Liu E, Hashemi M, Swersky K, et al. An imitation learning approach for cache replacement. In: Proceedings of the International Conference on Machine Learning, 2020. 6237\u20136247"},{"key":"3406_CR12","doi-asserted-by":"publisher","first-page":"134","DOI":"10.1145\/301464.301487","volume":"27","author":"D H Lee","year":"1999","unstructured":"Lee D H, Choi J, Kim J H, et al. On the existence of a spectrum of policies that subsumes the least recently used (LRU) and least frequently used (LFU) policies. SIGMETRICS Perform Eval Rev, 1999, 27: 134\u2013143","journal-title":"SIGMETRICS Perform Eval Rev"},{"key":"3406_CR13","first-page":"2151","volume":"3","author":"D Swain","year":"2011","unstructured":"Swain D, Paikaray B, Swain D. AWRP: adaptive weight ranking policy for improving cache performance. Comput Sci, 2011, 3: 2151\u20139617","journal-title":"Comput Sci"},{"key":"3406_CR14","doi-asserted-by":"crossref","unstructured":"Yu Y H, Wang W, Zhang J, et al. LERC: coordinated cache management for data-parallel systems. In: Proceedings of Global Communications Conference, 2017. 1\u20136","DOI":"10.1109\/GLOCOM.2017.8254999"},{"key":"3406_CR15","doi-asserted-by":"crossref","unstructured":"Zhao Y X, Wu J. Dache: a data aware caching for big-data applications using the MapReduce framework. In: Proceedings of International Conference on Computer Communications, 2013. 35\u201339","DOI":"10.1109\/INFCOM.2013.6566730"},{"key":"3406_CR16","doi-asserted-by":"crossref","unstructured":"Yang Z Y, Jia D L, Ioannidis S, et al. Intermediate data caching optimization for multi-stage and parallel big data frameworks. In: Proceedings of International Conference on Cloud Computing, San Francisco, 2018. 277\u2013284","DOI":"10.1109\/CLOUD.2018.00042"},{"key":"3406_CR17","unstructured":"Gonzalez J E, Xin R S, Dave A, et al. GraphX: graph processing in a distributed dataflow framework. In: Proceedings of USENIX Symposium on Operating Systems Design and Implementation, Broomfield, 2014. 599\u2013613"},{"key":"3406_CR18","first-page":"1235","volume":"17","author":"X R Meng","year":"2016","unstructured":"Meng X R, Bradley J, Yavuz B, et al. MLlib: machine learning in Apache Spark. J Mach Learn Res, 2016, 17: 1235\u20131241","journal-title":"J Mach Learn Res"},{"key":"3406_CR19","doi-asserted-by":"publisher","first-page":"399","DOI":"10.1016\/j.jss.2017.03.013","volume":"137","author":"L J Xu","year":"2018","unstructured":"Xu L J, Dou W S, Zhu F, et al. Characterizing and diagnosing out of memory errors in MapReduce applications. J Syst Softw, 2018, 137: 399\u2013414","journal-title":"J Syst Softw"},{"key":"3406_CR20","unstructured":"Ousterhout K, Rasti R, Ratnasamy S, et al. Making sense of performance in data analytics frameworks. In: Proceedings of USENIX Symposium on Networked Systems Design and implementation, Oakland, 2015. 293\u2013307"},{"key":"3406_CR21","doi-asserted-by":"crossref","unstructured":"Xu L, Li M, Zhang L, et al. MemTune: dynamic memory management for in-memory data analytic platforms. In: Proceedings of International Parallel and Distributed Processing Symposium, Chicago, 2016. 383\u2013392","DOI":"10.1109\/IPDPS.2016.105"},{"key":"3406_CR22","doi-asserted-by":"crossref","unstructured":"Li S, Amin M T, Ganti R, et al. Stark: optimizing in-memory computing for dynamic dataset collections. In: Proceedings of International Conference on Distributed Computing System, Atlanta, 2017. 103\u2013114","DOI":"10.1109\/ICDCS.2017.143"},{"key":"3406_CR23","unstructured":"Ananthanarayanan G, Ghodsi A, Warfield A, et al. PACMan: coordinated memory caching for parallel jobs. In: Proceedings of Symposium on Networked Systems Design and Implementation, San Jose, 2012. 267\u2013280"},{"key":"3406_CR24","doi-asserted-by":"crossref","unstructured":"Perez T B, Zhou X B, Chen D Z. Reference-distance eviction and prefetching for cache management in Spark. In: Proceedings of International Conference on Parallel Processing, Eugene, 2018. 1\u201310","DOI":"10.1145\/3225058.3225087"},{"key":"3406_CR25","unstructured":"Xu E, Saxena M, Chiu L. Neutrino: revisiting memory caching for iterative data analytics. In: Proceedings of USENIX Workshop on Hot Topics in Storage and File Systems, Denver, 2016. 16\u201320"}],"container-title":["Science China Information Sciences"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11432-021-3406-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11432-021-3406-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11432-021-3406-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,2,19]],"date-time":"2024-02-19T21:27:35Z","timestamp":1708378055000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11432-021-3406-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,12,26]]},"references-count":25,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2023,1]]}},"alternative-id":["3406"],"URL":"https:\/\/doi.org\/10.1007\/s11432-021-3406-5","relation":{},"ISSN":["1674-733X","1869-1919"],"issn-type":[{"value":"1674-733X","type":"print"},{"value":"1869-1919","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,12,26]]},"assertion":[{"value":"27 April 2021","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 November 2021","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 December 2021","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 December 2022","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"112104"}}