{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2023,6,20]],"date-time":"2023-06-20T17:10:30Z","timestamp":1687281030069},"reference-count":22,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2012,2,25]],"date-time":"2012-02-25T00:00:00Z","timestamp":1330128000000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Sci. China Inf. Sci."],"published-print":{"date-parts":[[2012,3]]},"DOI":"10.1007\/s11432-011-4497-z","type":"journal-article","created":{"date-parts":[[2012,2,24]],"date-time":"2012-02-24T06:17:05Z","timestamp":1330064225000},"page":"663-676","source":"Crossref","is-referenced-by-count":6,"title":["CUDA-Zero: a framework for porting shared memory GPU applications to multi-GPUs"],"prefix":"10.1007","volume":"55","author":[{"given":"DeHao","family":"Chen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"WenGuang","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"WeiMin","family":"Zheng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2012,2,25]]},"reference":[{"key":"4497_CR1","first-page":"1","volume-title":"SC\u201908: Proceedings of the 2008 ACM\/IEEE Conference on Supercomputing","author":"J. C. Phillips","year":"2008","unstructured":"Phillips J C, Stone J E, Schulten K. Adapting a message-driven parallel application to gpu-accelerated clusters. In: SC\u201908: Proceedings of the 2008 ACM\/IEEE Conference on Supercomputing. Piscataway: IEEE Press, 2008. 1\u20139"},{"key":"4497_CR2","doi-asserted-by":"crossref","first-page":"73","DOI":"10.1145\/1345206.1345220","volume-title":"PPoPP\u201908: Proceedings of the 13th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming","author":"S. Ryoo","year":"2008","unstructured":"Ryoo S, Rodrigues C I, Baghsorkhi S S, et al. Optimization principles and application performance evaluation of a multithreaded gpu using cuda. In: PPoPP\u201908: Proceedings of the 13th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming. New York: ACM, 2008. 73\u201382"},{"key":"4497_CR3","unstructured":"NVIDIA. NVIDIA CUDA Programming Guide 2.0. 2008"},{"key":"4497_CR4","doi-asserted-by":"crossref","first-page":"66","DOI":"10.1109\/MCSE.2010.69","volume":"12","author":"J. E. Stone","year":"2010","unstructured":"Stone J E, Gohara D, Shi G. OpenCL: a parallel programming standard for heterogeneous computing systems. Comput Sci Eng, 2010, 12: 66\u201373","journal-title":"Comput Sci Eng"},{"key":"4497_CR5","doi-asserted-by":"crossref","first-page":"777","DOI":"10.1145\/1186562.1015800","volume-title":"SIGGRAPH\u201904: ACM SIGGRAPH 2004 Papers","author":"I. Buck","year":"2004","unstructured":"Buck I, Foley T, Horn D, et al. Brook for gpus: stream computing on graphics hardware. In: SIGGRAPH\u201904: ACM SIGGRAPH 2004 Papers. New York: ACM, 2004. 777\u2013786"},{"key":"4497_CR6","doi-asserted-by":"crossref","first-page":"121","DOI":"10.1145\/1504176.1504196","volume-title":"PPoPP\u201909: Proceedings of the 14th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming","author":"G. Quintana Orti","year":"2009","unstructured":"Quintana Orti G, Igual F D, Quintana Orti E S, et al. Solving dense linear systems on platforms with multiple hardware accelerators. In: PPoPP\u201909: Proceedings of the 14th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming. New York: ACM, 2009. 121\u2013130"},{"key":"4497_CR7","first-page":"1","volume-title":"IPDPS\u201909: Proceedings of the 24th IEEE International Parallel and Distributed Processing Symposium","author":"S. Dana","year":"2009","unstructured":"Dana S, David K. Exploring the multi-gpu design space. In: IPDPS\u201909: Proceedings of the 24th IEEE International Parallel and Distributed Processing Symposium. New York: ACM, 2009. 1\u201312"},{"key":"4497_CR8","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1109\/IPDPS.2009.5161039","volume-title":"IEEE International Parallel and Distributed Processing Symposium","author":"N. Sundaram","year":"2009","unstructured":"Sundaram N, Raghunathan A, Chakradhar S T. A framework for efficient and scalable execution of domain-specific templates on gpus. In: IEEE International Parallel and Distributed Processing Symposium. Washington DC: IEEE, 2009. 1\u201312"},{"key":"4497_CR9","doi-asserted-by":"crossref","first-page":"31","DOI":"10.1145\/1283900.1283905","volume-title":"GH\u201906: Proceedings of the 21st ACM SIGGRAPH\/EUROGRAPHICS Symposium on Graphics Hardware","author":"A. Moerschell","year":"2006","unstructured":"Moerschell A, Owens J D. Distributed texture memory in a multi-gpu environment. In: GH\u201906: Proceedings of the 21st ACM SIGGRAPH\/EUROGRAPHICS Symposium on Graphics Hardware. New York: ACM, 2006. 31\u201338"},{"key":"4497_CR10","first-page":"1","volume-title":"GPGPU Workshop at Supercomputing","author":"F. Zhe","year":"2009","unstructured":"Zhe F, Feng Q, Arie K. Zippygpu: programming toolkit for general-purpose computation on gpu clusters. In: GPGPU Workshop at Supercomputing. Washington DC: IEEE, 2009. 1\u201312"},{"key":"4497_CR11","doi-asserted-by":"crossref","first-page":"605","DOI":"10.1109\/TVCG.2008.188","volume":"15","author":"M. Strengert","year":"2009","unstructured":"Strengert M, Muler C, Dachsbacher C, et al. CUDASA: compute unified device and systems architecture. IEEE Trans Vis Comput Gr, 2009, 15: 605\u2013617","journal-title":"IEEE Trans Vis Comput Gr"},{"key":"4497_CR12","doi-asserted-by":"crossref","first-page":"277","DOI":"10.1145\/1941553.1941591","volume-title":"Proceedings of the 16th ACM symposium on Principles and Practice of Parallel Programming","author":"J. Kim","year":"2011","unstructured":"Kim J, Kim H, Lee J H, et al. Achieving a single compute device image in opencl for multiple gpus. In: Proceedings of the 16th ACM symposium on Principles and Practice of Parallel Programming. New York: ACM, 2011. 277\u2013288"},{"key":"4497_CR13","first-page":"1","volume-title":"Compiling crystal for distributed-memory machines","author":"J. Li","year":"1992","unstructured":"Li J. Compiling crystal for distributed-memory machines. PhD Thesis. New Haven: Yale University, 1992. 1\u2013134"},{"key":"4497_CR14","first-page":"865","volume-title":"SC\u201990: Proceedings of the 1990 Conference on Supercomputing","author":"J. Li","year":"1990","unstructured":"Li J, Chen M. Generating explicit communication from shared-memory program references. In: SC\u201990: Proceedings of the 1990 Conference on Supercomputing. Los Alamitos: IEEE Computer Society Press, 1990. 865\u2013876"},{"key":"4497_CR15","doi-asserted-by":"crossref","first-page":"179","DOI":"10.1109\/71.127259","volume":"3","author":"M. Gupta","year":"1992","unstructured":"Gupta M, Banerjee P. Demonstration of automatic data partitioning techniques for parallelizing compilers on multicom puters. IEEE Trans Parall Distr, 1992, 3: 179\u2013193","journal-title":"IEEE Trans Parall Distr"},{"key":"4497_CR16","series-title":"LCPC","doi-asserted-by":"crossref","first-page":"16","DOI":"10.1007\/978-3-540-89740-8_2","volume-title":"Languages and Compilers for Parallel Computing: 21th International Workshop","author":"J. A. Stratton","year":"2008","unstructured":"Stratton J A, Stone S S, Hwu W M W. Mcuda: an efficient implementation of cuda kernels for multi-core cpus. In: Languages and Compilers for Parallel Computing: 21th International Workshop, LCPC 2008. New York: ACM, 2008. 16\u201330"},{"key":"4497_CR17","doi-asserted-by":"crossref","first-page":"610","DOI":"10.1145\/169627.169808","volume-title":"SC\u201993: Proceedings of the 1993 ACM\/IEEE Conference on Supercomputing","author":"A. Choudhary","year":"1993","unstructured":"Choudhary A, Koelbel C, Zosel M. High performance fortran: implementor and users workshop. In: SC\u201993: Proceedings of the 1993 ACM\/IEEE Conference on Supercomputing. New York: ACM, 1993. 610\u2013613"},{"key":"4497_CR18","doi-asserted-by":"crossref","first-page":"197","DOI":"10.1109\/32.842947","volume":"26","author":"B. L. Chamberlain","year":"2000","unstructured":"Chamberlain B L, Choi S E, Lewis E C, et al. Zpl: a machine independent programming language for parallel computers. IEEE Trans Software Eng, 2000, 26: 197\u2013211","journal-title":"IEEE Trans Software Eng"},{"key":"4497_CR19","doi-asserted-by":"crossref","first-page":"291","DOI":"10.1177\/1094342007078442","volume":"21","author":"B. Chamberlain","year":"2007","unstructured":"Chamberlain B, Callahan D, Zima H. Parallel programmability and the chapel language. Int J High Perform C, 2007, 21: 291\u2013312","journal-title":"Int J High Perform C"},{"key":"4497_CR20","doi-asserted-by":"crossref","first-page":"217","DOI":"10.1145\/1854273.1854303","volume-title":"Proceedings of the 19th International Conference on Parallel Architectures and Compilation Techniques","author":"C. Hong","year":"2010","unstructured":"Hong C, Chen D, Chen W, et al. Mapcg: writing parallel program portable between cpu and gpu. In: Proceedings of the 19th International Conference on Parallel Architectures and Compilation Techniques. New York: ACM, 2010. 217\u2013226"},{"key":"4497_CR21","first-page":"20","volume-title":"Proceedings of the EPHAM09 Workshop","author":"D. Chen","year":"2009","unstructured":"Chen D, Hong C, Chen W, et al. A mapreduce framework in heterogenous gpu environment. In: Proceedings of the EPHAM09 Workshop. New York: ACM, 2009. 20\u201327"},{"key":"4497_CR22","unstructured":"Impact research group. The parboil benchmark suite. http:\/\/www.crhc.uiuc.edu\/IMPACT\/parboil.php"}],"container-title":["Science China Information Sciences"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11432-011-4497-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11432-011-4497-z\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11432-011-4497-z","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,6,20]],"date-time":"2023-06-20T16:36:45Z","timestamp":1687279005000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11432-011-4497-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2012,2,25]]},"references-count":22,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2012,3]]}},"alternative-id":["4497"],"URL":"https:\/\/doi.org\/10.1007\/s11432-011-4497-z","relation":{},"ISSN":["1674-733X","1869-1919"],"issn-type":[{"value":"1674-733X","type":"print"},{"value":"1869-1919","type":"electronic"}],"subject":[],"published":{"date-parts":[[2012,2,25]]}}}