{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,5]],"date-time":"2025-10-05T19:44:19Z","timestamp":1759693459298},"reference-count":38,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2014,5,1]],"date-time":"2014-05-01T00:00:00Z","timestamp":1398902400000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J. Comput. Sci. Technol."],"published-print":{"date-parts":[[2014,5]]},"DOI":"10.1007\/s11390-014-1447-4","type":"journal-article","created":{"date-parts":[[2014,5,16]],"date-time":"2014-05-16T02:42:27Z","timestamp":1400208147000},"page":"532-546","update-policy":"http:\/\/dx.doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":13,"title":["OpenMC: Towards Simplifying Programming for TianHe Supercomputers"],"prefix":"10.1007","volume":"29","author":[{"given":"Xiang-Ke","family":"Liao","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Can-Qun","family":"Yung","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tao","family":"Tang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hui-Zhan","family":"Yi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Feng","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qiang","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jingling","family":"Xue","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2014,5,17]]},"reference":[{"issue":"3","key":"1447_CR1","doi-asserted-by":"crossref","first-page":"80","DOI":"10.1111\/j.1467-8659.2007.01012.x","volume":"26","author":"J Owens","year":"2007","unstructured":"Owens J, Luebke D, Govindaraju N et al. A survey of general purpose computation on graphics hardware. Computer Graphics Forum, 2007, 26(3): 80-113.","journal-title":"Computer Graphics Forum"},{"key":"1447_CR2","doi-asserted-by":"crossref","unstructured":"Sherlekar S. Tutorial: Intel many integrated core (MIC) architecture. In Proc. the 18th ICPADS, Dec. 2012, p.947.","DOI":"10.1109\/ICPADS.2012.162"},{"issue":"4","key":"1447_CR3","doi-asserted-by":"crossref","first-page":"445","DOI":"10.1007\/s11704-010-0383-x","volume":"4","author":"X Yang","year":"2010","unstructured":"Yang X, Liao X, Xu W et al. TH-1: China\u2019s first petaflop supercomputer. Frontiers of Computer Science in China, 2010, 4(4): 445-455.","journal-title":"Frontiers of Computer Science in China"},{"issue":"3","key":"1447_CR4","doi-asserted-by":"crossref","first-page":"344","DOI":"10.1007\/s02011-011-1137-8","volume":"26","author":"X Yang","year":"2011","unstructured":"Yang X, Liao X, Lu K et al. The TianHe-1A supercomputer: Its hardware and software. Journal of Computer Science and Technology, 2011, 26(3): 344-351.","journal-title":"Journal of Computer Science and Technology"},{"key":"1447_CR5","doi-asserted-by":"crossref","unstructured":"Kirk D. NVIDIA CUDA software and GPU parallel computing architecture. In Proc. International Symposium on Memory Management, Oct. 2007, pp.103-104.","DOI":"10.1145\/1296907.1296909"},{"key":"1447_CR6","unstructured":"Gaster B, Howes L, Kaeli D et al. Heterogeneous Computing with OpenCL \u2014 Revised OpenCL 1.2 Edition. Morgan Kaufmann, 2013."},{"key":"1447_CR7","doi-asserted-by":"crossref","unstructured":"Lee S, Vetter J. Early evaluation of directive-based GPU programming models for productive exascale computing. In Proc. Int. Conf. High Performance Computing, Networking, Storage and Analysis, Nov. 2012, Article No.23.","DOI":"10.1109\/SC.2012.51"},{"key":"1447_CR8","doi-asserted-by":"crossref","unstructured":"Wienke S, Springer P, Terboven C et al. OpenACC: First experiences with real-world applications. In Proc. the 18th Int. Conf. Euro-Par Parallel Processing, Aug. 2012, pp.859-870.","DOI":"10.1007\/978-3-642-32820-6_85"},{"key":"1447_CR9","doi-asserted-by":"crossref","unstructured":"Chapman B, Gropp W, Kumaran K et al. (eds.). OpenMP in the Petascale Era Springer, 2011.","DOI":"10.1007\/978-3-642-21487-5"},{"key":"1447_CR10","unstructured":"Petitet A, Whaley R, Dongarra J et al. HPL \u2014 A portable implementation of the high-performance linpack benchmark for distributed-memory computers, Sept. 2008. http:\/\/www.netlib.org\/benchmark\/hpl\/ , Mar. 2014."},{"issue":"7","key":"1447_CR11","doi-asserted-by":"crossref","first-page":"28","DOI":"10.1109\/2.869367","volume":"33","author":"J Henning","year":"2000","unstructured":"Henning J. SPEC CPU2000: Measuring CPU performance in the new millennium. Computer, 2000, 33(7): 28-35.","journal-title":"Computer"},{"issue":"1","key":"1447_CR12","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1006\/jcph.1995.1039","volume":"117","author":"S Plimpton","year":"1995","unstructured":"Plimpton S. Fast parallel algorithms for short-range molecular dynamics. J. Computational Physics, 1995, 117(1): 1-19.","journal-title":"J. Computational Physics"},{"key":"1447_CR13","unstructured":"Zhang A, Mo Z. Parallelization of lared-p codes for simulation of laser plasma interactions. Technical Report, ZW-J-2002045, Institute of Applied Physics and Computational Mathematics, 2002."},{"key":"1447_CR14","doi-asserted-by":"crossref","unstructured":"Kim J, Seo S, Lee J et al. SnuCL: An OpenCL framework for heterogeneous CPU\/GPU clusters. In Proc. the 26th ACM Int. Conf. Supercomputing, Jun. 2012, pp.341-352.","DOI":"10.1145\/2304576.2304623"},{"key":"1447_CR15","doi-asserted-by":"crossref","unstructured":"Cui H, Wang L, Xue J et al. Automatic library generation for BLAS3 on GPUs. In Proc. IEEE Int. Parallel and Distributed Processing Symposium, May 2011, pp.255-265.","DOI":"10.1109\/IPDPS.2011.33"},{"key":"1447_CR16","doi-asserted-by":"crossref","unstructured":"Di P, Wan Q, Zhang X et al. Toward harnessing DOACROSS parallelism for multi-GPGPUs. In Proc. the 39th Int. Conf. Parallel Processing, Sept. 2010, pp.40-50.","DOI":"10.1109\/ICPP.2010.13"},{"key":"1447_CR17","doi-asserted-by":"crossref","unstructured":"Di P, Xue J. Model-driven tile size selection for DOACROSS loops on GPUs. In Proc. 2011 Int. Conf. Euro-Par Parallel Processing, Aug. 2011, pp.401-412.","DOI":"10.1007\/978-3-642-23397-5_40"},{"key":"1447_CR18","doi-asserted-by":"crossref","unstructured":"Diogo M, Grelck C. Towards heterogeneous computing without heterogeneous programming. In Proc. the 13th Int. Symp. Trends in Functional Programming, June 2012, pp.279-294.","DOI":"10.1007\/978-3-642-40447-4_18"},{"key":"1447_CR19","doi-asserted-by":"crossref","unstructured":"Baskaran M, Ramanujam J, Sadayappan P. Automatic C-to-CUDA code generation for affine programs. In Proc. the 19th Int. Conf. Compiler Construction, Mar. 2010, pp.244-263.","DOI":"10.1007\/978-3-642-11970-5_14"},{"key":"1447_CR20","doi-asserted-by":"crossref","unstructured":"Cunningham D, Bordawekar R, Saraswat V. GPU programming in a high level language: Compiling X10 to CUDA. In Proc. the 2011 ACM SIGPLAN X10 Workshop, Jun. 2011, Article No.8.","DOI":"10.1145\/2212736.2212744"},{"key":"1447_CR21","doi-asserted-by":"crossref","unstructured":"Ohshima S, Hirasawa S, Honda H. OMPCUDA: OpenMP execution framework for CUDA based on Omni OpenMP compiler. In Proc. the 6th Int. Workshop. OpenMP, June 2010, pp.161-173.","DOI":"10.1007\/978-3-642-13217-9_13"},{"key":"1447_CR22","doi-asserted-by":"crossref","unstructured":"Lee S, Min S, Eigenmann R. OpenMP to GPGPU: A compiler framework for automatic translation and optimization. In Proc. the 14th PPoPP, Feb. 2009, pp.101-110.","DOI":"10.1145\/1504176.1504194"},{"key":"1447_CR23","doi-asserted-by":"crossref","unstructured":"Lee S, Eigenmann R. OpenMPC: Extended OpenMP programming and tuning for GPUs. In Proc. the 2010 ACM\/IEEE Int. Conf. High Performance Computing, Networking, Storage and Analysis, Nov. 2010, pp.1-11.","DOI":"10.1109\/SC.2010.36"},{"key":"1447_CR24","doi-asserted-by":"crossref","unstructured":"Hormati A, Samadi M,Woh M et al. Sponge: Portable stream programming on graphics engines. In Proc. the 16th Int. Conf. Architectural Support for Programming Languages and Operating Systems, Mar. 2011, pp.381-392.","DOI":"10.1145\/1961296.1950409"},{"issue":"6","key":"1447_CR25","doi-asserted-by":"crossref","first-page":"86","DOI":"10.1145\/1809028.1806606","volume":"45","author":"Y Yang","year":"2010","unstructured":"Yang Y, Xiang P, Kong J et al. A GPGPU compiler for memory optimization and parallelism management. ACM SIG-PLAN Notices, 2010, 45(6): 86-97.","journal-title":"ACM SIG-PLAN Notices"},{"key":"1447_CR26","doi-asserted-by":"crossref","unstructured":"Wu B, Zhao Z, Zhang E et al. Complexity analysis and algorithm design for reorganizing data to minimize non-coalesced memory accesses on GPU. In Proc. the 18th PPoPP, Feb. 2013, pp.57-68.","DOI":"10.1145\/2442516.2442523"},{"key":"1447_CR27","doi-asserted-by":"crossref","unstructured":"Reyes R, Lopez I, Fumero J et al. accull: An user-directed approach to heterogeneous programming. In Proc. IEEE the 10th ISPA, Jul. 2012, pp.654-661.","DOI":"10.1109\/ISPA.2012.97"},{"key":"1447_CR28","doi-asserted-by":"crossref","unstructured":"Han T, Abdelrahman T. hiCUDA: A high-level directive-based language for GPU programming. In Proc. the 2nd Workshop on General Purpose Processing on Graphics Processing Units, Mar. 2009, pp.52-61.","DOI":"10.1145\/1513895.1513902"},{"issue":"2","key":"1447_CR29","doi-asserted-by":"crossref","first-page":"173","DOI":"10.1142\/S0129626411000151","volume":"21","author":"A Duran","year":"2011","unstructured":"Duran A, Ayguad\u00e9 E, Badia R et al. OmpSs: A proposal for programming heterogeneous multi-core architectures. Parallel Processing Letters, 2011, 21(2): 173-193.","journal-title":"Parallel Processing Letters"},{"key":"1447_CR30","doi-asserted-by":"crossref","unstructured":"Auerbach J, Bacon D, Burcea I et al. A compiler and run-time for heterogeneous computing. In Proc. the 49th Annual Conference on Design Automation, Jun. 2012, pp.271-276.","DOI":"10.1145\/2228360.2228411"},{"key":"1447_CR31","doi-asserted-by":"crossref","unstructured":"Dubach C, Cheng P, Rabbah R et al. Compiling a high-level language for GPUs: (Via language support for architectures and compilers). In Proc. the 33rd PLDI, Jun. 2012, pp.1-12.","DOI":"10.1145\/2254064.2254066"},{"key":"1447_CR32","doi-asserted-by":"crossref","unstructured":"Cooper P, Dolinsky U, Donaldson A et al. Offload-automating code migration to heterogeneous multicore systems. In Proc. the 5th HiPEAC, Jan. 2010, pp.337-352.","DOI":"10.1007\/978-3-642-11515-8_25"},{"key":"1447_CR33","doi-asserted-by":"crossref","unstructured":"Beyer J, Stotzer E, Hart A et al. OpenMP for accelerators. In Proc. the 7th Int. Conf. OpenMP in the Petascale Era, June 2011, pp.108-121.","DOI":"10.1007\/978-3-642-21487-5_9"},{"key":"1447_CR34","doi-asserted-by":"crossref","unstructured":"UPC Consortium. UPC language specifications v1.2. Technical Report LBNL-59208, Lawrence Berkeley National Lab, 2005. http:\/\/upc.gwu.edu\/docs\/upc_specs_1.2.pdf , Mar. 2014.","DOI":"10.2172\/862127"},{"key":"1447_CR35","unstructured":"Saraswat V, Bloom B, Peshansky I et al. X10 language specification version 2.4. Technical Report, IBM, January 2012, http:\/\/x10.sourceforge.net\/documentation\/languagespec\/x-10-latest.pdf , Mar. 2014."},{"issue":"3","key":"1447_CR36","doi-asserted-by":"crossref","first-page":"291","DOI":"10.1177\/1094342007078442","volume":"21","author":"B Chamberlain","year":"2007","unstructured":"Chamberlain B, Callahan D, Zima H. Parallel programmability and the Chapel language. International Journal of High Performance Computing Applications, 2007, 21(3): 291-312.","journal-title":"International Journal of High Performance Computing Applications"},{"key":"1447_CR37","unstructured":"Hwu W W. GPU Computing Gems Jade Edition. Morgan Kaufmann, 2011."},{"key":"1447_CR38","doi-asserted-by":"crossref","unstructured":"Garland M, Kudlur M, Zheng Y. Designing a unified programming model for heterogeneous machines. In Proc. the International Conference on High Performance Computing, Networking, Storage and Analysis, Nov. 2012, Article No.67.","DOI":"10.1109\/SC.2012.48"}],"container-title":["Journal of Computer Science and Technology"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11390-014-1447-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11390-014-1447-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11390-014-1447-4","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,8,10]],"date-time":"2019-08-10T14:23:54Z","timestamp":1565447034000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11390-014-1447-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2014,5]]},"references-count":38,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2014,5]]}},"alternative-id":["1447"],"URL":"https:\/\/doi.org\/10.1007\/s11390-014-1447-4","relation":{},"ISSN":["1000-9000","1860-4749"],"issn-type":[{"value":"1000-9000","type":"print"},{"value":"1860-4749","type":"electronic"}],"subject":[],"published":{"date-parts":[[2014,5]]}}}