{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,9,4]],"date-time":"2024-09-04T13:50:52Z","timestamp":1725457852313},"publisher-location":"Berlin, Heidelberg","reference-count":21,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"type":"print","value":"9783642360350"},{"type":"electronic","value":"9783642360367"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2013]]},"DOI":"10.1007\/978-3-642-36036-7_12","type":"book-chapter","created":{"date-parts":[[2013,1,17]],"date-time":"2013-01-17T01:59:30Z","timestamp":1358387970000},"page":"171-184","source":"Crossref","is-referenced-by-count":0,"title":["Fine-Grained Treatment to Synchronizations in GPU-to-CPU Translation"],"prefix":"10.1007","author":[{"given":"Ziyu","family":"Guo","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xipeng","family":"Shen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"key":"12_CR1","unstructured":"Hpcgpu project, \n                  \n                    http:\/\/hpcgpu.codeplex.com\/"},{"key":"12_CR2","unstructured":"NVIDIA CUDA Programming Guide, \n                  \n                    http:\/\/developer.download.nvidia.com"},{"key":"12_CR3","unstructured":"OpenCL, \n                  \n                    http:\/\/www.khronos.org\/opencl\/"},{"key":"12_CR4","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"154","DOI":"10.1007\/978-3-642-02303-3_13","volume-title":"Evolving OpenMP in an Age of Extreme Parallelism","author":"E. Ayguade","year":"2009","unstructured":"Ayguade, E., Badia, R.M., Cabrera, D., Duran, A., Gonzalez, M., Igual, F., Jimenez, D., Labarta, J., Martorell, X., Mayo, R., Perez, J.M., Quintana-Ort\u00ed, E.S.: A Proposal to Extend the OpenMP Tasking Model for Heterogeneous Architectures. In: M\u00fcller, M.S., de Supinski, B.R., Chapman, B.M. (eds.) IWOMP 2009. LNCS, vol.\u00a05568, pp. 154\u2013167. Springer, Heidelberg (2009)"},{"key":"12_CR5","doi-asserted-by":"crossref","unstructured":"Baskaran, M.M., Bondhugula, U., Krishnamoorthy, S., Ramanujam, J., Rountev, A., Sadayappan, P.: A compiler framework for optimization of affine loop nests for GPGPUs. In: ICS 2008: Proceedings of the 22nd Annual International Conference on Supercomputing, pp. 225\u2013234 (2008)","DOI":"10.1145\/1375527.1375562"},{"key":"12_CR6","doi-asserted-by":"crossref","unstructured":"Carrillo, S., Siegel, J., Li, X.: A control-structure splitting optimization for GPGPU. In: Proceedings of ACM Computing Frontiers (2009)","DOI":"10.1145\/1531743.1531766"},{"key":"12_CR7","doi-asserted-by":"crossref","unstructured":"Cooper, K., Torczon, L.: Engineering a Compiler. Morgan Kaufmann (2003)","DOI":"10.1016\/B0-12-227410-5\/00126-5"},{"key":"12_CR8","doi-asserted-by":"crossref","unstructured":"Diamos, G., Kerr, A., Yalamanchili, S., Clark, N.: Ocelot: A dynamic compiler for bulk-synchronous applications in heterogeneous systems. In: Proceedings of the Nineteenth International Conference on Parallel Architectures and Compilation Techniques. ACM (2010)","DOI":"10.1145\/1854273.1854318"},{"key":"12_CR9","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"16","DOI":"10.1007\/978-3-540-89740-8_2","volume-title":"Languages and Compilers for Parallel Computing","author":"J.A. Stratton","year":"2008","unstructured":"Stratton, J.A., Stone, S.S., Hwu, W.-M.W.: MCUDA: An Efficient Implementation of CUDA Kernels for Multi-core CPUs. In: Amaral, J.N. (ed.) LCPC 2008. LNCS, vol.\u00a05335, pp. 16\u201330. Springer, Heidelberg (2008)"},{"key":"12_CR10","doi-asserted-by":"crossref","unstructured":"Stratton, J.A., et al.: Efficient compilation of fine-grained SPMD-threadedprograms for multicore CPUs. In: CGO 2010 (2010)","DOI":"10.1145\/1772954.1772971"},{"key":"12_CR11","first-page":"407","volume-title":"MICRO 2007: Proceedings of the 40th Annual IEEE\/ACM International Symposium on Microarchitecture","author":"W. Fung","year":"2007","unstructured":"Fung, W., Sham, I., Yuan, G., Aamodt, T.: Dynamic warp formation and scheduling for efficient GPU control flow. In: MICRO 2007: Proceedings of the 40th Annual IEEE\/ACM International Symposium on Microarchitecture, pp. 407\u2013420. IEEE Computer Society, Washington, DC (2007)"},{"key":"12_CR12","doi-asserted-by":"crossref","unstructured":"Guo, Z., Zhang, E., Shen, X.: Correctly treating synchronizations in compiling fine-grained SPMD-threaded programs for CPU. In: Proceedings of International Conference on Parallel Architectures and Compilation Techniques (2011)","DOI":"10.1109\/PACT.2011.62"},{"key":"12_CR13","doi-asserted-by":"crossref","unstructured":"Hormati, A., Samadi, M., Woh, M., Mudge, T., Mahlke, S.: Sponge: Portable stream programming on graphics engines. In: ASPLOS 2011 (2011)","DOI":"10.1145\/1950365.1950409"},{"key":"12_CR14","doi-asserted-by":"crossref","unstructured":"Lee, S., Min, S.-J., Eigenmann, R.: Openmp to GPGPU: a compiler framework for automatic translation and optimization. In: PPOPP 2009, pp. 101\u2013110 (2009)","DOI":"10.1145\/1594835.1504194"},{"key":"12_CR15","doi-asserted-by":"crossref","unstructured":"Meng, J., Tarjan, D., Skadron, K.: Dynamic warp subdivision for integrated branch and memory divergence tolerance. In: ISCA 2010 (2010)","DOI":"10.1145\/1815961.1815992"},{"key":"12_CR16","unstructured":"Michel, S., Philipp, K., Sergei, G.: Skelcl - a portable skeleton library for high-level GPU programming. In: IPDPS 2011 (2011)"},{"key":"12_CR17","doi-asserted-by":"crossref","unstructured":"Ryoo, S., Rodrigues, C.I., Baghsorkhi, S.S., Stone, S.S., Kirk, D.B., Hwu, W.W.: Optimization principles and application performance evaluation of a multithreaded GPU using CUDA. In: PPoPP 2008: Proceedings of the 13th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming, pp. 73\u201382 (2008)","DOI":"10.1145\/1345206.1345220"},{"key":"12_CR18","doi-asserted-by":"crossref","unstructured":"Tarjan, D., Meng, J., Skadron, K.: Increasing memory latency tolerance for SIMD cores. In: SC 2009 (2009)","DOI":"10.1145\/1654059.1654082"},{"key":"12_CR19","doi-asserted-by":"crossref","unstructured":"Yang, Y., Xiang, P., Kong, J., Zhou, H.: A GPGPU compiler for memory optimization and parallelism management. In: PLDI (2010)","DOI":"10.1145\/1806596.1806606"},{"key":"12_CR20","doi-asserted-by":"crossref","unstructured":"Zhang, E.Z., Jiang, Y., Guo, Z., Shen, X.: Streamlining GPU applications on the fly. In: Proceedings of the ACM International Conference on Supercomputing, ICS, pp. 115\u2013125 (2010)","DOI":"10.1145\/1810085.1810104"},{"key":"12_CR21","doi-asserted-by":"crossref","unstructured":"Zhang, E.Z., Jiang, Y., Guo, Z., Tian, K., Shen, X.: On-the-fly elimination of dynamic irregularities for GPU computing. In: ASPLOS 2011 (2011)","DOI":"10.1145\/1950365.1950408"}],"container-title":["Lecture Notes in Computer Science","Languages and Compilers for Parallel Computing"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-642-36036-7_12.pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2021,5,4]],"date-time":"2021-05-04T13:34:31Z","timestamp":1620135271000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-642-36036-7_12"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2013]]},"ISBN":["9783642360350","9783642360367"],"references-count":21,"URL":"https:\/\/doi.org\/10.1007\/978-3-642-36036-7_12","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2013]]}}}