{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,29]],"date-time":"2025-06-29T04:40:03Z","timestamp":1751172003793,"version":"3.41.0"},"reference-count":48,"publisher":"Springer Science and Business Media LLC","issue":"S1","license":[{"start":{"date-parts":[[2017,12,22]],"date-time":"2017-12-22T00:00:00Z","timestamp":1513900800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"funder":[{"name":"Postdoctoral Science Foundation of Chin","award":["2013T61007","2013M542468"],"award-info":[{"award-number":["2013T61007","2013M542468"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Cluster Comput"],"published-print":{"date-parts":[[2019,1]]},"DOI":"10.1007\/s10586-017-1295-4","type":"journal-article","created":{"date-parts":[[2017,12,22]],"date-time":"2017-12-22T16:53:38Z","timestamp":1513961618000},"page":"2129-2144","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["PRODA: improving parallel programs on GPUs through dependency analysis"],"prefix":"10.1007","volume":"22","author":[{"given":"Xiong","family":"Wei","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ming","family":"Hu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tao","family":"Peng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Minghua","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhiying","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiao","family":"Qin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2017,12,22]]},"reference":[{"issue":"1","key":"1295_CR1","doi-asserted-by":"publisher","first-page":"108","DOI":"10.1109\/JPROC.2008.2007472","volume":"97","author":"P Jacob","year":"2009","unstructured":"Jacob, P., Zia, A., Erdogan, O., Belemjian, P.M., Kim, J.W., Chu, M., Kraft, R.P., Mcdonald, J.F., Bernstein, K.: Mitigating memory wall effects in high-clock-rate and multicore cmos 3-d processor memory stacks. Proc. IEEE 97(1), 108\u2013122 (2009)","journal-title":"Proc. IEEE"},{"issue":"2","key":"1295_CR2","doi-asserted-by":"publisher","first-page":"39","DOI":"10.1109\/MM.2008.31","volume":"28","author":"E Lindholm","year":"2008","unstructured":"Lindholm, E., Nickolls, J., Oberman, S., Montrym, J.: Nvidia tesla: a unified graphics and computing architecture. IEEE Micro 28(2), 39\u201355 (2008)","journal-title":"IEEE Micro"},{"key":"1295_CR3","unstructured":"Hennessy, J.L., Patterson, D.A., Arpaci-Dusseau, A.C.: Computer Architecture: A Quantitative Approach. Morgan Kaufmann Pub., an imprint of Elsevier (2007)"},{"key":"1295_CR4","doi-asserted-by":"crossref","unstructured":"Koop, M.J., Huang, W., Gopalakrishnan, K., Panda, D.K.: Performance analysis and evaluation of PCIE 2.0 and quad-data rate infiniband. In: Proceedings of the 2008 16th IEEE Symposium on High Performance Interconnects, pp. 85\u201392 (2008)","DOI":"10.1109\/HOTI.2008.26"},{"key":"1295_CR5","doi-asserted-by":"crossref","unstructured":"Stone, J.E., Gohara, D., Shi, G.: Opencl: a parallel programming standard for heterogeneous computing systems. In: IEEE Des. Test, pp. 66\u201373 (2010)","DOI":"10.1109\/MCSE.2010.69"},{"key":"1295_CR6","unstructured":"Pacheco, P.S.: An Introduction to Parallel Programming, Vol. 5, No. 4, p. 357359 (2011)"},{"issue":"8","key":"1295_CR7","first-page":"1132","volume":"24","author":"LI Jian-Minga","year":"2009","unstructured":"Jian-Minga, L.I., Xiang-Peib, H.U., Pang, Z.L., Qian, K.M.: A parallel ant colony optimization algorithm based on fine-grained model with gpu-accelerated. Control Decis. 24(8), 1132\u20131136 (2009)","journal-title":"Control Decis."},{"key":"1295_CR8","doi-asserted-by":"crossref","unstructured":"Mohr, E., Kranz, D.A., Halstead, R.H. and Jr.: Lazy task creation: a technique for increasing the granularity of parallel programs. In: IEEE Transactions on Parallel and Distributed Systems, pp. 264\u2013280 (1991)","DOI":"10.1109\/71.86103"},{"issue":"12","key":"1295_CR9","doi-asserted-by":"publisher","first-page":"4135","DOI":"10.1021\/ct2005193","volume":"7","author":"BG Levine","year":"2011","unstructured":"Levine, B.G., Lebard, D.N., Devane, R., Shinoda, W., Kohlmeyer, A., Klein, M.L.: Micellization studied by gpu-accelerated coarse-grained molecular dynamics. J. Chem. Theory Comput. 7(12), 4135\u20134145 (2011)","journal-title":"J. Chem. Theory Comput."},{"key":"1295_CR10","volume-title":"Parallel Programming\u2014for Multicore and Cluster Systems","author":"T Rauber","year":"2010","unstructured":"Rauber, T., Rnger, G.: Parallel Programming\u2014for Multicore and Cluster Systems. Springer, Heidelberg (2010)"},{"key":"1295_CR11","unstructured":"Hwu, W.M., Ryoo, S., Ueng, S.Z., Kelm, J.H., Gelado, I., Stone, S.S., Kidd, R.E., Baghsorkhi, S.S., Mahesri, A.A., Tsao, S.C.: Implicitly parallel programming models for thousand-core microprocessors. In: Design Automation Conference, 2007. DAC \u201907. 44th ACM\/IEEE, pp. 754\u2013759 (2007)"},{"key":"1295_CR12","doi-asserted-by":"crossref","unstructured":"Lucas, P.: The development of the data-parallel gpu programming language CGIS. In: In International Conference on Computational Science, pp. 200\u2013203 (2006)","DOI":"10.1007\/11758549_31"},{"key":"1295_CR13","unstructured":"Mellorcrummey, J.: Center for programming models for scalable parallel computing. In: Scitech Connect Center for Programming Models for Scalable Parallel Computing (2008)"},{"key":"1295_CR14","doi-asserted-by":"crossref","unstructured":"Bikshandi, G., Guo, J., Hoeflinger, D., Almsi, G., Fraguela, B.B., Garzarn, M.J., Padua, D.A., Praun, C.V.: Programming for parallelism and locality with hierarchically tiled arrays. In: Proceedings of the Eleventh Acm Sigplan Symposium on Principles and Practice of Parallel Program, pp. 48\u201357 (2006)","DOI":"10.1145\/1122971.1122981"},{"key":"1295_CR15","doi-asserted-by":"crossref","unstructured":"D\u2019Alberto, P.D., Nicolau, A.: Adaptive Strassen\u2019s matrix multiplication. In: ICs Proceedings of Annual International Conference on Supercomputing, pp. 284\u2013292 (2007)","DOI":"10.1145\/1274971.1275010"},{"issue":"6","key":"1295_CR16","doi-asserted-by":"publisher","first-page":"2080","DOI":"10.1007\/s11227-014-1333-4","volume":"72","author":"Z Wang","year":"2016","unstructured":"Wang, Z., Liu, Y., Chiu, S.: An efficient parallel collaborative filtering algorithm on multi-gpu platform. J. Supercomput. 72(6), 2080\u20132094 (2016)","journal-title":"J. Supercomput."},{"key":"1295_CR17","doi-asserted-by":"crossref","unstructured":"Cui, S., Gro\u00dfsch\u00e4dl, J., Liu, Z., Xu, Q.: High-speed elliptic curve cryptography on the NVIDIA GT200 graphics processing unit. In: Lecture Notes in Computer Science (2014)","DOI":"10.1007\/978-3-319-06320-1_16"},{"issue":"6","key":"1295_CR18","doi-asserted-by":"publisher","first-page":"16581664","DOI":"10.1002\/mrm.22112","volume":"62","author":"S Roujol","year":"2009","unstructured":"Roujol, S., De Senneville, B.D., Vahala, E., S\u00f8rensen, T.S., Moonen, C., Ries, M.: Online real-time reconstruction of adaptive TSENSE with commodity CPU\/GPU hardware. Magn. Reson. Med. 62(6), 16581664 (2009)","journal-title":"Magn. Reson. Med."},{"key":"1295_CR19","unstructured":"Tetsuya, O., Minh, T.T., Jinpil, L., Taisuke, B., Mitsuhisa, S.: Extend to GPU for Xcalablemp: a parallel programming language. In: IPSJ Sig. Notes (2011)"},{"key":"1295_CR20","unstructured":"Choi, W.H., Liu, X.: Case study: runtime reduction of a buffer insertion algorithm using GPU parallel programming. In: SOC Conference (SOCC), 2010 IEEE International, pp. 121\u2013126 (2010)"},{"key":"1295_CR21","unstructured":"Raymond, N., Samuel, T., Olivier, A.: GPU\/CPU Work Sharing Mechanism on XMP-dev, High-level Parallel Programming Language for GPU Cluster, Vol. 2014, pp. 87\u201396 (2013)"},{"issue":"2","key":"1295_CR22","doi-asserted-by":"publisher","first-page":"28","DOI":"10.1109\/MM.2012.2","volume":"32","author":"A Branover","year":"2012","unstructured":"Branover, A., Foley, D., Steinman, M.: Amd fusion apu: Llano. IEEE Micro 32(2), 28\u201337 (2012)","journal-title":"IEEE Micro"},{"issue":"4","key":"1295_CR23","doi-asserted-by":"publisher","first-page":"501","DOI":"10.1145\/4472.4478","volume":"7","author":"RHH Jr","year":"1985","unstructured":"Jr, R.H.H.: Multilisp: a language for concurrent symbolic computation. ACM Trans. Program. Lang. Syst. 7(4), 501\u2013538 (1985)","journal-title":"ACM Trans. Program. Lang. Syst."},{"key":"1295_CR24","doi-asserted-by":"crossref","unstructured":"Zhang, C., Huang, K., Cui, X., Chen, Y.: Programming-level power measurement for GPU clusters. In: Green Computing and Communications (GreenCom). IEEE\/ACM International Conference on, Vol. 2011, pp. 182\u2013187 (2011)","DOI":"10.1109\/GreenCom.2011.38"},{"key":"1295_CR25","unstructured":"Wataru, T., Xu, J., Ken, W.: An implementation and evaluation of a compiler for ACTGPU, an actor-based asynchronous parallel programming language. In: IPSJ Sig Notes, vol. 2012 (2012)"},{"issue":"12","key":"1295_CR26","doi-asserted-by":"publisher","first-page":"147","DOI":"10.1016\/S0304-3975(00)00051-7","volume":"248","author":"B Grant","year":"2000","unstructured":"Grant, B., Mock, M., Philipose, M., Chambers, C., Eggers, S.J.: DyC: an expressive annotation-directed dynamic compiler for c. Theor. Comput. Sci. 248(12), 147\u2013199 (2000)","journal-title":"Theor. Comput. Sci."},{"key":"1295_CR27","doi-asserted-by":"crossref","unstructured":"Maruyama, N., Nomura, T., Sato, K., Matsuoka, S.: Physis: an implicitly parallel programming model for stencil computations on large-scale GPU-accelerated supercomputers. In: Proceedings of 2011 International Conference for High Performance Computing, Networking, Storage and Analysis, pp. 1\u201312 (2011)","DOI":"10.1145\/2063384.2063398"},{"key":"1295_CR28","doi-asserted-by":"crossref","unstructured":"Lattner, C., Adve, V.: Llvm: a compilation framework for lifelong program analysis and transformation. In: Proceedings of the international symposium on Code generation and optimization: feedback-directed and runtime optimization, pp. 75\u201386 (2004)","DOI":"10.1109\/CGO.2004.1281665"},{"key":"1295_CR29","doi-asserted-by":"crossref","unstructured":"Kerr, A., Diamos, G., Yalamanchili, S.: A characterization and analysis of PTX kernels. In: Workload Characterization, 2009. IISWC 2009. IEEE International Symposium on, pp. 3\u201312 (2009)","DOI":"10.1109\/IISWC.2009.5306801"},{"key":"1295_CR30","doi-asserted-by":"crossref","unstructured":"Chang, C.T., Chen, Y.S., Wu, I.W., Shann, J.J.: A translation framework for automatic translation of annotated llvm ir into opencl kernel function. In: Smart Innovation Systems and Technologies (2013)","DOI":"10.1007\/978-3-642-35473-1_62"},{"issue":"5","key":"1295_CR31","doi-asserted-by":"publisher","first-page":"1688","DOI":"10.1007\/s11661-011-0993-4","volume":"43","author":"A Saeed-Akbari","year":"2012","unstructured":"Saeed-Akbari, A., Mosecker, L., Schwedt, A., Bleck, W.: Characterization and prediction of flow behavior in high-manganese twinning induced plasticity steels: Part I. Mechanism maps and work-hardening behavior. Metall. Mater. Trans. A 43(5), 1688\u20131704 (2012)","journal-title":"Metall. Mater. Trans. A"},{"key":"1295_CR32","doi-asserted-by":"crossref","unstructured":"Lee, J., Sato, M., Boku, T.: Openmpd: a directive-based data parallel language extension for distributed memory systems pp. 121\u2013128 (2008)","DOI":"10.1109\/ICPP-W.2008.28"},{"issue":"3","key":"1295_CR33","first-page":"421439","volume":"49","author":"F Wolf","year":"2003","unstructured":"Wolf, F., Mohr, B.: Automatic performance analysis of hybrid MPI\/OpenMP applications. J. Syst. Archit. 49(3), 421439 (2003)","journal-title":"J. Syst. Archit."},{"key":"1295_CR34","doi-asserted-by":"crossref","unstructured":"Linderman, M.D., Collins, J.D., Wang, H., Meng, T.H.: Merge: a programming model for heterogeneous multi-core systems. In: ASPLOS XIII: Proceedings of the 13th International Conference on Architectural, pp. 287\u2013296 (2008)","DOI":"10.1145\/1346281.1346318"},{"key":"1295_CR35","doi-asserted-by":"publisher","first-page":"197220","DOI":"10.1016\/j.jpdc.2005.08.002","volume":"66","author":"A Lastovetsky","year":"2006","unstructured":"Lastovetsky, A., Reddy, R.: Heterompi: Towards a message-passing library for heterogeneous networks of computers. Journal of Parallel and Distributed Computing 66, 197220 (2006)","journal-title":"Journal of Parallel and Distributed Computing"},{"issue":"3\u20134","key":"1295_CR36","first-page":"211","volume":"29","author":"M Knobloch","year":"2013","unstructured":"Knobloch, M., Foszczynski, M., Homberg, W., Pleiter, D., Bttiger, H.: Mapping fine-grained power measurements to HPC application runtime characteristics on IBM POWER7. Comput. Sci. Res. Dev. 29(3\u20134), 211\u2013219 (2013)","journal-title":"Comput. Sci. Res. Dev."},{"key":"1295_CR37","unstructured":"Hoshi, T., Ootsu, K., Ohkawa, T., and Yokota, T.: \u201cRuntime overhead reduction in automated parallel processing system using valgrind,\u201d in International Symposium on Computing and NETWORKING, (2013) pp. 572\u2013576"},{"key":"1295_CR38","unstructured":"Guire, N.M.: Linux kernel GCOV-tool analysis (2006)"},{"key":"1295_CR39","doi-asserted-by":"crossref","unstructured":"Wang, G., Tang, T., Fang, X., Ren, X.: Program optimization of array-intensive spec2k benchmarks on multithreaded GPU using CUDA and brook+. In: Parallel and Distributed Systems (ICPADS), 2009 15th International Conference on, pp. 292\u2013299 (2009)","DOI":"10.1109\/ICPADS.2009.12"},{"key":"1295_CR40","doi-asserted-by":"crossref","unstructured":"Hong, S., Kim, H.: An analytical model for a GPU architecture with memory-level and thread-level parallelism awareness. In: ISCA \u201909 Proceedings of the 36th Annual International Symposium on Computer Architecture, pp. 152\u2013163 (2009)","DOI":"10.1145\/1555754.1555775"},{"issue":"1","key":"1295_CR41","doi-asserted-by":"publisher","first-page":"131","DOI":"10.1007\/s10586-011-0179-2","volume":"16","author":"W Ma","year":"2013","unstructured":"Ma, W., Krishnamoorthy, S., Villa, O., Kowalski, K., Agrawal, G.: Optimizing tensor contraction expressions for hybrid cpu-gpu execution. Clust. Comput. 16(1), 131\u2013155 (2013)","journal-title":"Clust. Comput."},{"key":"1295_CR42","volume-title":"GPU Accelerated Curve Fitting with IDL","author":"M Galloy","year":"2012","unstructured":"Galloy, M.: GPU Accelerated Curve Fitting with IDL. American Geophysical Union, Washington, DC (2012)"},{"issue":"1","key":"1295_CR43","doi-asserted-by":"publisher","first-page":"39","DOI":"10.1142\/S0129626406002459","volume":"16","author":"T Nakashima","year":"2006","unstructured":"Nakashima, T., Fujiwara, A.: A cost optimal parallel algorithm for patience sorting. Parallel Process. Lett. 16(1), 39\u201351 (2006)","journal-title":"Parallel Process. Lett."},{"issue":"3","key":"1295_CR44","doi-asserted-by":"publisher","first-page":"271","DOI":"10.1007\/BF02240073","volume":"36","author":"SG Akl","year":"1986","unstructured":"Akl, S.G.: An adaptive and cost-optimal parallel algorithm for minimum spanning trees. Computing 36(3), 271\u2013277 (1986)","journal-title":"Computing"},{"issue":"3","key":"1295_CR45","first-page":"297","volume":"3","author":"P Alonso","year":"2003","unstructured":"Alonso, P., Cortina, R., Daz, I., Hernndez, V., Ranilla, J.: A simple cost-optimal parallel algorithm to solve linear equation systems. Information 3(3), 297\u2013304 (2003)","journal-title":"Information"},{"key":"1295_CR46","doi-asserted-by":"crossref","unstructured":"Bahl, A.K., Baltzer, O., Rau-Chaplin, A., Varghese, B., Whiteway, A.: Multi-GPU computing for achieving speedup in real-time aggregate risk analysis. High performance computing on graphics processing units (hgpu.org, Chaplin, 2013)","DOI":"10.1109\/ICPP.2013.108"},{"key":"1295_CR47","unstructured":"Zhao, X.D., Liang, S.X., Sun, Z.C., Liu, Z.B., Han, S.L., Ren, X.F.: Foundation and analysis of computational efficiency for hydrodynamic model based on GPU parallel algorithm. J. Dalian Univ. Technol. (2014)"},{"key":"1295_CR48","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-662-04722-4","volume-title":"The Design of Rijndael: AES the Advanced Encryption Standard","author":"J Daemen","year":"2002","unstructured":"Daemen, J., Rijmen, V.: The Design of Rijndael: AES the Advanced Encryption Standard. Springer, Berlin (2002)"}],"container-title":["Cluster Computing"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10586-017-1295-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10586-017-1295-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10586-017-1295-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,29]],"date-time":"2025-06-29T04:23:55Z","timestamp":1751171035000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10586-017-1295-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017,12,22]]},"references-count":48,"journal-issue":{"issue":"S1","published-print":{"date-parts":[[2019,1]]}},"alternative-id":["1295"],"URL":"https:\/\/doi.org\/10.1007\/s10586-017-1295-4","relation":{},"ISSN":["1386-7857","1573-7543"],"issn-type":[{"type":"print","value":"1386-7857"},{"type":"electronic","value":"1573-7543"}],"subject":[],"published":{"date-parts":[[2017,12,22]]},"assertion":[{"value":"14 January 2017","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 September 2017","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 October 2017","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 December 2017","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}