{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,27]],"date-time":"2025-03-27T10:16:10Z","timestamp":1743070570645,"version":"3.40.3"},"publisher-location":"Berlin, Heidelberg","reference-count":33,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"type":"print","value":"9783662588338"},{"type":"electronic","value":"9783662588345"}],"license":[{"start":{"date-parts":[[2019,1,1]],"date-time":"2019-01-01T00:00:00Z","timestamp":1546300800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2019]]},"DOI":"10.1007\/978-3-662-58834-5_2","type":"book-chapter","created":{"date-parts":[[2019,2,22]],"date-time":"2019-02-22T07:12:05Z","timestamp":1550819525000},"page":"21-38","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Programmable and Scalable Architecture for Graphics Processing Units"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7251-636X","authenticated-orcid":false,"given":"Carlos S.","family":"de La Lama","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5707-8544","authenticated-orcid":false,"given":"Pekka","family":"J\u00e4\u00e4skel\u00e4inen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1845-2924","authenticated-orcid":false,"given":"Heikki","family":"Kultala","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0097-1010","authenticated-orcid":false,"given":"Jarmo","family":"Takala","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2019,2,23]]},"reference":[{"key":"2_CR1","doi-asserted-by":"publisher","unstructured":"Colwell, R.P., Nix, R.P., O\u2019Donnell, J.J., Papworth, D.B., Rodman, P.K.: A VLIW architecture for a trace scheduling compiler. In: Proceedings of 2nd International Conference on Architectural Support for Programming Languages and Operating Systems, pp. 180\u2013192. IEEE Computer Society Press, Los Alamitos (1987). \n                    https:\/\/doi.org\/10.1145\/36206.36201","DOI":"10.1145\/36206.36201"},{"key":"2_CR2","volume-title":"Microprocessor Architectures: From VLIW to TTA","author":"H Corporaal","year":"1997","unstructured":"Corporaal, H.: Microprocessor Architectures: From VLIW to TTA. Wiley, Chichester (1997)"},{"issue":"12\u201313","key":"2_CR3","doi-asserted-by":"publisher","first-page":"949","DOI":"10.1016\/S1383-7621(98)00046-0","volume":"45","author":"H Corporaal","year":"1999","unstructured":"Corporaal, H.: TTAs: missing the ILP complexity wall. J. Syst. Arch.: EUROMICRO J. 45(12\u201313), 949\u2013973 (1999). \n                    https:\/\/doi.org\/10.1016\/S1383-7621(98)00046-0","journal-title":"J. Syst. Arch.: EUROMICRO J."},{"key":"2_CR4","unstructured":"Crow, T.S.: Evolution of the graphical processing unit. Master\u2019s thesis, University of Nevada, Reno (2004)"},{"issue":"10","key":"2_CR5","doi-asserted-by":"publisher","first-page":"50","DOI":"10.1145\/1400181.1400197","volume":"51","author":"K Fatahalian","year":"2008","unstructured":"Fatahalian, K., Houston, M.: A closer look at GPUs. Commun. ACM 51(10), 50\u201357 (2008)","journal-title":"Commun. ACM"},{"key":"2_CR6","unstructured":"Halfhill, T.R.: Parallel processing with CUDA. Microprocessor Report (2008)"},{"key":"2_CR7","doi-asserted-by":"publisher","unstructured":"Hoogerbrugge, J., Corporaal, H.: Register file port requirements of transport triggered architectures. In: Proceedings of 27th International Symposium on Microarchitecture, pp. 191\u2013195. ACM, New York (1994). \n                    https:\/\/doi.org\/10.1145\/192724.192751","DOI":"10.1145\/192724.192751"},{"key":"2_CR8","doi-asserted-by":"crossref","unstructured":"J\u00e4\u00e4skel\u00e4inen, P., Guzma, V., Cilio, A., Takala, J.: Codesign toolset for application-specific instruction-set processors. In: Proceedings of SPIE Multimedia on Mobile Devices 2007, vol. 6507 (2007)","DOI":"10.1117\/12.707233"},{"key":"2_CR9","doi-asserted-by":"crossref","unstructured":"J\u00e4\u00e4skel\u00e4inen, P., de La Lama, C.S., Huerta, P., Takala, J.: OpenCL-based design methodology for application-specific processors. In: 10th International Conference on Embedded Computer Systems: Architectures, Modeling and Simulation, July 2010, to appear","DOI":"10.1109\/ICSAMOS.2010.5642061"},{"key":"2_CR10","unstructured":"Kessenich, J.: The OpenGL Shading Language. 3DLabs, Inc. (2006)"},{"key":"2_CR11","unstructured":"Khronos Group: OpenCL 1.0 Specification (2009). \n                    http:\/\/www.khronos.org\/registry\/cl\/"},{"key":"2_CR12","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"2","DOI":"10.1007\/978-3-642-03138-0_2","volume-title":"Embedded Computer Systems: Architectures, Modeling, and Simulation","author":"CS Lama de La","year":"2009","unstructured":"de La Lama, C.S., J\u00e4\u00e4skel\u00e4inen, P., Takala, J.: Programmable and scalable architecture for graphics processing units. In: Bertels, K., Dimopoulos, N., Silvano, C., Wong, S. (eds.) SAMOS 2009. LNCS, vol. 5657, pp. 2\u201311. Springer, Heidelberg (2009). \n                    https:\/\/doi.org\/10.1007\/978-3-642-03138-0_2"},{"key":"2_CR13","unstructured":"Lattner, C., Adve, V.: LLVM: a compilation framework for lifelong program analysis & transformation. In: Proceedings of International Symposium on Code Generation and Optimization: Feedback-Directed and Runtime Optimization. IEEE Computer Society, Washington (2004)"},{"issue":"2","key":"2_CR14","doi-asserted-by":"publisher","first-page":"39","DOI":"10.1109\/MM.2008.31","volume":"28","author":"E Lindholm","year":"2008","unstructured":"Lindholm, E., Nickolls, J., Oberman, S., Montrym, J.: NVIDIA Tesla: a unified graphics and computing architecture. IEEE Micro 28(2), 39\u201355 (2008)","journal-title":"IEEE Micro"},{"key":"2_CR15","unstructured":"Lorie, R.A., Hovey R. Strong, J.: Method for conditional branch execution in SIMD vector processors. US Patent 4435758 (1984)"},{"issue":"2","key":"2_CR16","doi-asserted-by":"publisher","first-page":"96","DOI":"10.1109\/MC.2007.59","volume":"40","author":"D Luebke","year":"2007","unstructured":"Luebke, D., Humphreys, G.: How GPUs work. Computer 40(2), 96\u2013100 (2007)","journal-title":"Computer"},{"key":"2_CR17","unstructured":"Moy, S., Lindholm, J.E.: Method and system for programmable pipelined graphics processing with branching instructions. US Patent 6947047 (2005)"},{"key":"2_CR18","unstructured":"Moya, V., Gonz\u00e1lez, C., Roca, J., Fern\u00e1ndez, A., Espasa, R.: Shader performance analysis on a modern GPU architecture. In: Proceedings of 38th IEEE\/ACM International Symposium on Microarchitecture, pp. 355\u2013364 (2005)"},{"issue":"2","key":"2_CR19","doi-asserted-by":"publisher","first-page":"56","DOI":"10.1109\/MM.2010.41","volume":"30","author":"J Nickolls","year":"2010","unstructured":"Nickolls, J., Dally, W.: The GPU computing era. IEEE Micro 30(2), 56\u201369 (2010). \n                    https:\/\/doi.org\/10.1109\/MM.2010.41","journal-title":"IEEE Micro"},{"key":"2_CR20","unstructured":"NVIDIA: CUDA programming guide v2.1. Technical report (2008)"},{"key":"2_CR21","unstructured":"NVIDIA: NVIDIA\u2019s next generation CUDA compute architecture: Fermi. White Paper (2009)"},{"issue":"5","key":"2_CR22","doi-asserted-by":"publisher","first-page":"879","DOI":"10.1109\/JPROC.2008.917757","volume":"96","author":"JD Owens","year":"2008","unstructured":"Owens, J.D., Houston, M., Luebke, D., Green, S., Stone, J.E., Phillips, J.C.: GPU computing. Proc. IEEE 96(5), 879\u2013899 (2008)","journal-title":"Proc. IEEE"},{"issue":"1","key":"2_CR23","doi-asserted-by":"publisher","first-page":"80","DOI":"10.1111\/j.1467-8659.2007.01012.x","volume":"26","author":"JD Owens","year":"2007","unstructured":"Owens, J.D., Luebke, D., Govindaraju, N., Harris, M., Kr\u00fcger, J., Lefohn, A.E., Purcell, T.J.: A survey of general-purpose computation on graphics hardware. Comput. Graph. Forum 26(1), 80\u2013113 (2007)","journal-title":"Comput. Graph. Forum"},{"issue":"5","key":"2_CR24","doi-asserted-by":"publisher","first-page":"895","DOI":"10.1145\/330249.330250","volume":"21","author":"M Poletto","year":"1999","unstructured":"Poletto, M., Sarkar, V.: Linear scan register allocation. ACM T. Program. Lang. Syst. 21(5), 895\u2013913 (1999). \n                    https:\/\/doi.org\/10.1145\/330249.330250","journal-title":"ACM T. Program. Lang. Syst."},{"key":"2_CR25","volume-title":"OpenGL Shading Language","author":"RJ Rost","year":"2010","unstructured":"Rost, R.J.: OpenGL Shading Language, 3rd edn. Addison-Wesley, Reading (2010)","edition":"3"},{"key":"2_CR26","unstructured":"Segal, M., Akeley, K.: The OpenGL Graphics System: A Specification. Silicon Graphics, Inc. (2006)"},{"issue":"3","key":"2_CR27","doi-asserted-by":"publisher","first-page":"18","DOI":"10.1145\/1360612.1360617","volume":"27","author":"L Seiler","year":"2008","unstructured":"Seiler, L., et al.: Larrabee: a many-core x86 architecture for visual computing. ACM Trans. Graph. 27(3), 18 (2008)","journal-title":"ACM Trans. Graph."},{"key":"2_CR28","unstructured":"Smelyanskiy, M., Mahlke, S.A., Davidson, E.S., Lee, H.H.S.: Predicate-aware scheduling: a technique for reducing resource constraints. In: Proceedings of International Symposium on Code Generation and Optimization: Feedback-Directed and Runtime Optimization. ACM International Conference Proceedings Series, vol. 37, pp. 169\u2013178 (2003)"},{"key":"2_CR29","volume-title":"The Complete Effect and HLSL Guide","author":"S St-Laurent","year":"2005","unstructured":"St-Laurent, S.: The Complete Effect and HLSL Guide. Paradoxal Press, Redmond (2005)"},{"issue":"7","key":"2_CR30","doi-asserted-by":"publisher","first-page":"491","DOI":"10.1007\/s002360050095","volume":"34","author":"R Stephens","year":"1997","unstructured":"Stephens, R.: A survey of stream processing. Acta Inform. 34(7), 491\u2013541 (1997)","journal-title":"Acta Inform."},{"key":"2_CR31","unstructured":"Tampere University of Technology: TCE project at TUT. \n                    http:\/\/tce.cs.tut.fi"},{"key":"2_CR32","unstructured":"Wasson, S.: AMD Radeon HD 2900 XT graphics processor: R600 revealed. Technical report (2007). \n                    http:\/\/www.techreport.com\/reviews\/2007q2\/radeon-hd-2900xt\/index.x?pg=1"},{"key":"2_CR33","unstructured":"Wasson, S.: NVIDIA\u2019s GeForce 8800 graphics processor. Technical report (2007). \n                    http:\/\/www.techreport.com\/reviews\/2006q4\/geforce-8800\/index.x?pg=1"}],"container-title":["Lecture Notes in Computer Science","Transactions on High-Performance Embedded Architectures and Compilers V"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-662-58834-5_2","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,5,20]],"date-time":"2019-05-20T07:31:46Z","timestamp":1558337506000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-662-58834-5_2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019]]},"ISBN":["9783662588338","9783662588345"],"references-count":33,"URL":"https:\/\/doi.org\/10.1007\/978-3-662-58834-5_2","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2019]]},"assertion":[{"value":"23 February 2019","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}}]}}