{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,26]],"date-time":"2026-07-26T04:42:36Z","timestamp":1785040956536,"version":"3.55.0"},"publisher-location":"Berlin, Heidelberg","reference-count":26,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"value":"9783642288685","type":"print"},{"value":"9783642288692","type":"electronic"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2012]]},"DOI":"10.1007\/978-3-642-28869-2_16","type":"book-chapter","created":{"date-parts":[[2012,3,22]],"date-time":"2012-03-22T16:44:36Z","timestamp":1332434676000},"page":"316-335","source":"Crossref","is-referenced-by-count":31,"title":["On the Correctness of the SIMT Execution Model of GPUs"],"prefix":"10.1007","author":[{"given":"Axel","family":"Habermaier","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Alexander","family":"Knapp","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","reference":[{"key":"16_CR1","unstructured":"AMD. Evergreen Family Instruction Set Architecture, Reference Guide (2011)"},{"key":"16_CR2","doi-asserted-by":"crossref","unstructured":"Barnat, J., Brim, L., Ceska, M., Lamr, T.: CUDA Accelerated LTL Model Checking. In: Proc. 15th Int. Conf. Parallel and Distributed Systems (ICPADS 2009), pp. 34\u201341 (2009)","DOI":"10.1109\/ICPADS.2009.50"},{"key":"16_CR3","doi-asserted-by":"crossref","unstructured":"Bo\u0161na\u010dki, D., Edelkamp, S., Sulewski, D., Wijs, A.: GPU-PRISM: An Extension of PRISM for General Purpose Graphics Processing Units. In: Proc. 9th Int. Wsh. Parallel and Distributed Methods in Verification (PDMV 2010), pp. 17\u201319 (2010)","DOI":"10.1109\/PDMC-HiBi.2010.11"},{"key":"16_CR4","unstructured":"Collange, S.: Stack-less SIMT Reconvergence at Low Cost. Technical Report HAL-00622654, INRIA (2011)"},{"key":"16_CR5","unstructured":"Coon, B.W., Nickolls, J.R., Nyland, L., Mills, P.C., Lindholm, J.E.: Indirect Function Call Instructions in a Synchronous Parallel Thread Processor, United States Patent Application #2009\/0240931 (2009)"},{"key":"16_CR6","doi-asserted-by":"crossref","unstructured":"Fung, W.W.L., Aamodt, T.M.: Thread Block Compaction for Efficient SIMT Control Flow. In: Proc. 17th IEEE Int. Symp. High Performance Computer Architecture (HPCA 2011), pp. 25\u201336 (2011)","DOI":"10.1109\/HPCA.2011.5749714"},{"key":"16_CR7","doi-asserted-by":"crossref","unstructured":"Fung, W.W.L., Sham, I., Yuan, G., Aamodt, T.M.: Dynamic Warp Formation and Scheduling for Efficient GPU Control Flow. In: Proc. 40th Ann. IEEE\/ACM Int. Symp. Microarchitecture (MICRO 2007), pp. 407\u2013420 (2007)","DOI":"10.1109\/MICRO.2007.30"},{"issue":"4","key":"16_CR8","doi-asserted-by":"publisher","first-page":"13","DOI":"10.1109\/MM.2008.57","volume":"28","author":"M. Garland","year":"2008","unstructured":"Garland, M., Le Grand, S., Nickolls, J., Anderson, J., Hardwick, J., Morton, S., Phillips, E., Zhang, Y., Volkov, V.: Parallel Computing Experiences with CUDA. IEEE Micro\u00a028(4), 13\u201327 (2008)","journal-title":"IEEE Micro"},{"key":"16_CR9","unstructured":"Habermaier, A.: The Model of Computation of CUDA and its Formal Semantics. Technical Report 2011-14, University of Augsburg (2011)"},{"key":"16_CR10","doi-asserted-by":"crossref","unstructured":"Habermaier, A., Knapp, A.: On the Correctness of the SIMT Execution Model of GPUs. Technical Report 2012-1, University of Augsburg (2012)","DOI":"10.1007\/978-3-642-28869-2_16"},{"key":"16_CR11","unstructured":"Hennessy, J.L., Patterson, D.A.: Computer Architecture: A Quantitative Approach, 5th edn. Elsevier Science & Technology (2011)"},{"key":"16_CR12","unstructured":"Khronos Group Inc. The OpenGL Shading Language 4.20, Revision 6 (2011)"},{"key":"16_CR13","unstructured":"Khronos OpenCL Working Group. The OpenCL Specification 1.2, Revision 15 (2011)"},{"key":"16_CR14","doi-asserted-by":"publisher","first-page":"77","DOI":"10.1145\/964965.808581","volume":"18","author":"A. Levinthal","year":"1984","unstructured":"Levinthal, A., Porter, T.: Chap \u2013 A SIMD Graphics Processor. SIGGRAPH Comput. Graph.\u00a018, 77\u201382 (1984)","journal-title":"SIGGRAPH Comput. Graph."},{"key":"16_CR15","unstructured":"The LLVM Compiler Infrastructure, \n                  \n                    http:\/\/www.llvm.org\/\n                  \n                  \n                 (01\/04\/2012)"},{"key":"16_CR16","unstructured":"Mantor, M., Houston, M.: AMD Graphic Core Next: Low Power High Performance Graphics & Parallel Compute. Presentation at the AMD Fusion Developer Summit (2011)"},{"key":"16_CR17","doi-asserted-by":"publisher","first-page":"54","DOI":"10.1145\/1365490.1365501","volume":"6","author":"W. Mark","year":"2008","unstructured":"Mark, W.: Future Graphics Architectures. ACM Queue\u00a06, 54\u201364 (2008)","journal-title":"ACM Queue"},{"key":"16_CR18","doi-asserted-by":"crossref","unstructured":"Meng, J., Tarjan, D., Skadron, K.: Dynamic Warp Subdivision for Integrated Branch and Memory Divergence Tolerance. In: Proc. 37th Ann. Int. Symp. Computer Architecture (ISCA 2010), pp. 235\u2013246 (2010)","DOI":"10.1145\/1815961.1815992"},{"key":"16_CR19","unstructured":"Moy, S., Lindholm, J.E.: Method and System for Programmable Pipelined Graphics Processing with Branching Instructions, United States Patent #6,947,047 (2005)"},{"key":"16_CR20","unstructured":"Muchnick, S.S.: Advanced Compiler Design and Implementation. Morgan Kaufmann Publishers Inc. (1997)"},{"issue":"2","key":"16_CR21","doi-asserted-by":"publisher","first-page":"56","DOI":"10.1109\/MM.2010.41","volume":"30","author":"J.R. Nickolls","year":"2010","unstructured":"Nickolls, J.R., Dally, W.: The GPU Computing Era. IEEE Micro\u00a030(2), 56\u201369 (2010)","journal-title":"IEEE Micro"},{"key":"16_CR22","unstructured":"NVIDIA. DirectCompute Programming Guide 3.2 (2010)"},{"key":"16_CR23","unstructured":"NVIDIA. cuobjdump. CUDA Toolkit 4.1 (2011)"},{"key":"16_CR24","unstructured":"NVIDIA. NVIDIA CUDA C Programming Guide 4.1 (2011)"},{"key":"16_CR25","unstructured":"NVIDIA. NVIDIA Opens Up CUDA Platform by Releasing Compiler Source Code (2011), \n                  \n                    http:\/\/tiny.cc\/NvidiaLLVM\n                  \n                  \n                 (01\/04\/2012)"},{"key":"16_CR26","doi-asserted-by":"crossref","unstructured":"Reynolds, J.C.: Theories of Programming Languages. Cambridge University Press (1998)","DOI":"10.1017\/CBO9780511626364"}],"container-title":["Lecture Notes in Computer Science","Programming Languages and Systems"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-642-28869-2_16.pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2021,5,4]],"date-time":"2021-05-04T07:13:39Z","timestamp":1620112419000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-642-28869-2_16"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2012]]},"ISBN":["9783642288685","9783642288692"],"references-count":26,"URL":"https:\/\/doi.org\/10.1007\/978-3-642-28869-2_16","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2012]]}}}