{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,29]],"date-time":"2025-09-29T07:47:01Z","timestamp":1759132021398},"reference-count":35,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2011,10,29]],"date-time":"2011-10-29T00:00:00Z","timestamp":1319846400000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Cluster Comput"],"published-print":{"date-parts":[[2013,3]]},"DOI":"10.1007\/s10586-011-0188-1","type":"journal-article","created":{"date-parts":[[2011,10,28]],"date-time":"2011-10-28T15:59:14Z","timestamp":1319817554000},"page":"77-90","source":"Crossref","is-referenced-by-count":5,"title":["Energy cost evaluation of parallel algorithms for multiprocessor systems"],"prefix":"10.1007","volume":"16","author":[{"given":"Zhuowei","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xianbin","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Naixue","family":"Xiong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Laurence T.","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wuqing","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2011,10,29]]},"reference":[{"key":"188_CR1","unstructured":"NVIDIA Corporation: NVIDIA CUDA compute unfied device architecture programming guide. http:\/\/developer.nvidia.com\/cuda\/ (2011)"},{"key":"188_CR2","unstructured":"http:\/\/www.cise.urf.edu\/research\/sparse\/matrices\/ (2011)"},{"key":"188_CR3","first-page":"C-20","volume-title":"ACM Workshop on General-Purpose Computing on Graphics Processors (GP2)","author":"I. Buck","year":"2004","unstructured":"Buck, I., Fatahalian, K., Hanrahan, P.: GPUBench, Evaluating GPU performance for numerical and scientific applications. In: ACM Workshop on General-Purpose Computing on Graphics Processors (GP2), p.\u00a0C-20 (2004)"},{"key":"188_CR4","doi-asserted-by":"crossref","first-page":"105","DOI":"10.1145\/1693453.1693470","volume-title":"Proceedings of the 15th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming (PPoPP 2010)","author":"S.S. Baghsorkhi","year":"2010","unstructured":"Baghsorkhi, S.S., Delahaye, M., Patel, S.J., Gropp, W.D., Hwu, W.W.: An adaptive performance modeling tool for GPU architectures. In: Proceedings of the 15th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming (PPoPP 2010), pp.\u00a0105-114. ACM, New York (2010)"},{"key":"188_CR5","volume-title":"ACM\/IEEE SC","author":"B. He","year":"2007","unstructured":"He, B., et al.: Efficient gather and scatter operations on graphics processors. In: ACM\/IEEE SC (2007)"},{"issue":"10\u201311","key":"188_CR6","doi-asserted-by":"crossref","first-page":"685","DOI":"10.1016\/j.parco.2007.09.002","volume":"33","author":"D. Goddeke","year":"2007","unstructured":"Goddeke, D., et al.: Exploring weak scalability for FEM calculations on a GPU-enhanced cluster. Parallel Comput. 33(10\u201311), 685\u2013699 (2007)","journal-title":"Parallel Comput."},{"key":"188_CR7","volume-title":"SC","author":"N.K. Govindaraju","year":"2006","unstructured":"Govindaraju, N.K., Larsen, S., Gray, J., Manocha, D.: A memory model for scientific algorithms on graphics processors. In: SC (2006)"},{"key":"188_CR8","doi-asserted-by":"crossref","first-page":"306","DOI":"10.1109\/DSD.2005.40","volume-title":"The Eighth Euromicro Conference on Digital System Design, Architectures, Methods, and Tools","author":"P. Trancoso","year":"2005","unstructured":"Trancoso, P., Charalambous, M.: Exploring graphics processor performance for general purpose applications. In: The Eighth Euromicro Conference on Digital System Design, Architectures, Methods, and Tools, pp. 306\u2013313 (2005)"},{"key":"188_CR9","volume-title":"Parallel and Distributed Computing and Networks (PDCN)","author":"O. Harrison","year":"2007","unstructured":"Harrison, O., Waldron, J.: Optimising data movement rates for parallel processing applications on graphics processors. In: Parallel and Distributed Computing and Networks (PDCN) (2007)"},{"key":"188_CR10","volume-title":"IEEE International Symposium on Performance Analysis of Systems and Software","author":"J. Sheaffer","year":"2005","unstructured":"Sheaffer, J., Skadron, K., Luebke, D.: Studding thermal management for graphic-processor architectures. In: IEEE International Symposium on Performance Analysis of Systems and Software (2005)"},{"key":"188_CR11","volume-title":"Workshop on General Purpose Processing on Graphics Processing Units (GPGPU)","author":"K. Ramani","year":"2007","unstructured":"Ramani, K., Ibrahim, A., Shimizu, D.: PowerRed: a flexible power modeling frame work for power efficiency exploration in GPUs. In: Workshop on General Purpose Processing on Graphics Processing Units (GPGPU) (2007)"},{"key":"188_CR12","volume-title":"the Third international Workshop on Automatic Performance Tuning","author":"H. Tajuzawa","year":"2008","unstructured":"Tajuzawa, H., Satol, K., Kobay Ashi, H.: SPRAT: runtime processor selection for energy-aware computing. In: the Third international Workshop on Automatic Performance Tuning (2008)"},{"key":"188_CR13","volume-title":"Workshop on Power Aware Computing and System","author":"M. Rofouei","year":"2008","unstructured":"Rofouei, M., Stathopoulos, T., Ryffel, S., Kaiser, W., Sarrafzadeh, M.: Energy-aware high performance computing with graphic processing units. In: Workshop on Power Aware Computing and System (2008)"},{"key":"188_CR14","volume-title":"23rd IEEE International Parallel and Distributed Processing Symposium (IPDPS)","author":"S. Huang","year":"2009","unstructured":"Huang, S., Xiao, S., Feng, W.: On the energy efficiency of graphics processing units for scientific computing. In: 23rd IEEE International Parallel and Distributed Processing Symposium (IPDPS) (2009)"},{"key":"188_CR15","volume-title":"SPAA","author":"V.A. Korthikanti","year":"2010","unstructured":"Korthikanti, V.A., Agha, G.: Towards optimizing energy costs of algorithms for shared memory architectures. In: SPAA (2010)"},{"key":"188_CR16","first-page":"212","volume-title":"ICPP","author":"V.A. Korthikanti","year":"2009","unstructured":"Korthikanti, V.A., Agha, G.: Analysis of parallel algorithms for energy conservation in scalable multicore architectures. In: ICPP, pp. 212\u2013219 (2009)"},{"key":"188_CR17","doi-asserted-by":"crossref","first-page":"386","DOI":"10.1109\/ISQED.2007.119","volume-title":"International Symposium on Quality Electronic Design","author":"X. Wang","year":"2007","unstructured":"Wang, X., Ziavras, S.: Performance-energy tradeoff for matrix multiplication on FPGA-based mixed-mode chip multiprocessors. In: International Symposium on Quality Electronic Design, pp. 386\u2013391 (2007)"},{"key":"188_CR18","first-page":"228","volume-title":"SPAA, Parallel Computing","author":"M.A. Bender","year":"2005","unstructured":"Bender, M.A., Fineman, J.T.: Concurrent cache-oblibious b-trees. In: SPAA, Parallel Computing, pp. 228\u2013237 (2005). 18171616"},{"key":"188_CR19","doi-asserted-by":"crossref","first-page":"1116","DOI":"10.1145\/48529.48535","volume":"31","author":"A. Aggarwal","year":"1988","unstructured":"Aggarwal, A., Viteer, J.S.: The input\/output complexity of sorting and related problems. Commun. ACM 31, 1116\u20131127 (1988)","journal-title":"Commun. ACM"},{"issue":"4","key":"188_CR20","doi-asserted-by":"crossref","first-page":"473","DOI":"10.1109\/4.126534","volume":"27","author":"A. Chandrakasan","year":"1992","unstructured":"Chandrakasan, A., Sheng, S., Brodersen, R.: Low-power CMOS digital design. IEEE J. Solid-State Circuits 27(4), 473\u2013484 (1992)","journal-title":"IEEE J. Solid-State Circuits"},{"key":"188_CR21","volume-title":"Synthesis of Parallel Algorithms","author":"G.E. Blelloch","year":"1990","unstructured":"Blelloch, G.E.: Prefix sums and their applications. In: Reif, J.H. (ed.) Synthesis of Parallel Algorithms. Morgan Kaufmann, San Mateo (1990)"},{"key":"188_CR22","volume-title":"GPU Gems 3","author":"M. Harri","year":"2007","unstructured":"Harri, M., Sengupta, S., Owens, J.D.: Parallel prefix sum (scan) with CUDA. In: Nguyen, H. (ed.) GPU Gems 3. Addison-Wesley, Reading (2007)"},{"key":"188_CR23","first-page":"97","volume-title":"Graphics Hardware 2007","author":"S. Sengupta","year":"2007","unstructured":"Sengupta, S., Harris, M., Zhang, Y., Owens, J.D.: Scan primitives for GPU computing. In: Graphics Hardware 2007, pp. 97\u2013106. ACM Press, New York (2007)"},{"key":"188_CR24","unstructured":"Bell, N., Garland, M.: Efficient sparse matrix-vector multiplication on CUDA. NVIDIA Technical Report NVR-2008-004, NVIDIA Corporation (2008)"},{"issue":"3","key":"188_CR25","doi-asserted-by":"crossref","first-page":"917","DOI":"10.1145\/882262.882364","volume":"22","author":"J. Bolz","year":"2003","unstructured":"Bolz, J., Farmer, I., Grinspun, E., Schroder, P.: SPARSE matrix slovers on the GPU: Conjugate gradients and multigrid. ACM Trans. Graph. 22(3), 917\u2013924 (2003). Proceedings of ACM SIGGRAPH","journal-title":"ACM Trans. Graph."},{"key":"188_CR26","unstructured":"Blelloch, G.E., Heroux, M.A., Zagham: Segmented operations for sparse matrix computation on vector multiprocessors. Technical Report CMU-CS-93-173, School of Computer Science, Carnegie Mellon University, August 1993"},{"key":"188_CR27","unstructured":"Vazquez, F., Garzon, E.M., Martinez, J.A., Fernandex, J.J.: Scan primitives for vector computers. The sparse matrix vector produce on GPUs. Computer Architecture and Electronics Dep., University of Almeria (2009)"},{"key":"188_CR28","unstructured":"Baskaran, M.M., Bordawekar, R.: Optimizing Sparse Matrix-Vector Multiplication on GPUs. IBM Research Report RC24704 (2009)"},{"key":"188_CR29","doi-asserted-by":"crossref","first-page":"109","DOI":"10.1109\/71.485501","volume":"7","author":"A.J.C. Bik","year":"1996","unstructured":"Bik, A.J.C., Wijshoff, H.A.G.: Automatic data structure selection and transformation for sparse matrix computations. IEEE Trans. Parallel Distrib. Syst. 7, 109\u2013126 (1996)","journal-title":"IEEE Trans. Parallel Distrib. Syst."},{"key":"188_CR30","doi-asserted-by":"crossref","first-page":"205","DOI":"10.1145\/1375527.1375559","volume-title":"ICS: Proceedings of the 22nd Annual International Conference on Supercomputing","author":"Y. Dotesenko","year":"2008","unstructured":"Dotesenko, Y., Govindaraju, N.K., Sloan, P.-P., Boyd, C., Manferdelli, J.: Fast scan algorithms on graphics processors. In: ICS: Proceedings of the 22nd Annual International Conference on Supercomputing, New York, NY, USA, pp. 205\u2013213. ACM Press, New York (2008)"},{"key":"188_CR31","first-page":"666","volume-title":"Supercomputing\u201990: Proceedings of the 1990 Conference on Supercomputing","author":"S. Chatterhee","year":"1990","unstructured":"Chatterhee, S., Blelloch, G.E.: Zagham., Scan primitives for vector computers. In: Supercomputing\u201990: Proceedings of the 1990 Conference on Supercomputing, pp. 666\u2013675 (1990)"},{"key":"188_CR32","volume-title":"Proceedings of 19th Eurographics\/SIGGRAPH Graphics Hardware Workshop, Graphics Hardware","author":"K. Fatahaian","year":"2004","unstructured":"Fatahaian, K., Sugerman, J., Hanrahan, P.: Understanding the efficiency of GPU algorithms for matrix-matrix multiplications. In: Proceedings of 19th Eurographics\/SIGGRAPH Graphics Hardware Workshop, Graphics Hardware, Grenoble, France (2004)"},{"key":"188_CR33","volume-title":"ICETE: Proceeding of the 2nd International Conference on Education Technology and Computer","author":"Z. Wang","year":"2010","unstructured":"Wang, Z., Xu, X., Zhao, W., Zhang, Y., He, S.: Optimizing Sparse Matrix-Vector Multiplication on CUDA. In: ICETE: Proceeding of the 2nd International Conference on Education Technology and Computer, Shanghai, China (2010)"},{"key":"188_CR34","unstructured":"CUDDP: CUDA Data Parallel Primitives Library. http:\/\/www.gpgpu.org\/developer\/cudpp\/ (2011)"},{"key":"188_CR35","doi-asserted-by":"crossref","first-page":"9","DOI":"10.1145\/344166.344181","volume-title":"Proceeding of the 2000 International Symposium on Low Power Electronics and Design","author":"T. Burd","year":"2000","unstructured":"Burd, T., Brodersen, R.: Design issues for dynamic voltage scaling. In: Proceeding of the 2000 International Symposium on Low Power Electronics and Design, (ISLPED\u201900) Rapallo, Italy, pp. 9\u201314 (2000)"}],"container-title":["Cluster Computing"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10586-011-0188-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10586-011-0188-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10586-011-0188-1","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,5,30]],"date-time":"2019-05-30T14:40:14Z","timestamp":1559227214000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10586-011-0188-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2011,10,29]]},"references-count":35,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2013,3]]}},"alternative-id":["188"],"URL":"https:\/\/doi.org\/10.1007\/s10586-011-0188-1","relation":{},"ISSN":["1386-7857","1573-7543"],"issn-type":[{"value":"1386-7857","type":"print"},{"value":"1573-7543","type":"electronic"}],"subject":[],"published":{"date-parts":[[2011,10,29]]}}}