{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2022,3,30]],"date-time":"2022-03-30T20:10:03Z","timestamp":1648671003209},"reference-count":30,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2014,9,28]],"date-time":"2014-09-28T00:00:00Z","timestamp":1411862400000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Sign Process Syst"],"published-print":{"date-parts":[[2015,7]]},"DOI":"10.1007\/s11265-014-0957-1","type":"journal-article","created":{"date-parts":[[2014,9,27]],"date-time":"2014-09-27T02:04:29Z","timestamp":1411783469000},"page":"87-101","update-policy":"http:\/\/dx.doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["A Co-Design Framework with OpenCL Support for Low-Energy Wide SIMD Processor"],"prefix":"10.1007","volume":"80","author":[{"given":"Dongrui","family":"She","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yifan","family":"He","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Luc","family":"Waeijen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Henk","family":"Corporaal","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2014,9,28]]},"reference":[{"key":"957_CR1","unstructured":"Cadence: Tensilica Customizable Processor IP. URL http:\/\/ip.cadence.com\/ipportfolio\/tensilica-ip\/ ."},{"key":"957_CR2","unstructured":"Kyo, S., & Okazaki, S. (2008). IMAPCAR: A 100 GOPS In-Vehicle Vision Processor Based on 128 Ring Connected Four-Way VLIW Processing Elements. Journal of Signal Processing Systems, 1\u201312."},{"issue":"1","key":"957_CR3","doi-asserted-by":"crossref","first-page":"192","DOI":"10.1109\/JSSC.2007.909328","volume":"43","author":"A Abbo","year":"2008","unstructured":"Abbo, A., & et al. (2008). Xetal-II: a 107 GOPS, 600 mW massively parallel processor for video scene analysis. IEEE Journal of Solid-State Circuits, 43(1), 192\u2013201.","journal-title":"IEEE Journal of Solid-State Circuits"},{"key":"957_CR4","unstructured":"AMD: AMD OpenCL Zone. URL http:\/\/developer.amd.com\/resources\/heterogeneous-computing\/opencl-zone ."},{"key":"957_CR5","doi-asserted-by":"crossref","unstructured":"Lattner, C., & Adve, V. (2004). LLVM: A compilation framework for lifelong program analysis & transformation. In: Proceedings of the 2004 International Symposium on Code Generation and Optimization (CGO\u201904), pp. 75\u201386.","DOI":"10.1109\/CGO.2004.1281665"},{"issue":"2","key":"957_CR6","doi-asserted-by":"crossref","first-page":"50","DOI":"10.1109\/MM.2011.24","volume":"31","author":"C Wittenbrink","year":"2011","unstructured":"Wittenbrink, C., & et al. (2011). Fermi GF100 GPU architecture. IEEE Micro, 31(2), 50\u201359.","journal-title":"IEEE Micro"},{"key":"957_CR7","unstructured":"CACTI: cacti 5.3, rev 174. URL http:\/\/quid.hpl.hp.com:9081\/cacti\/ ."},{"key":"957_CR8","doi-asserted-by":"crossref","unstructured":"She, D., & et al. (2012). Energy efficient special instruction support in an embedded processor with compact isa. In: Proceedings of the 2012 International Conference on Compilers, Architecture, and Synthesis for Embedded Systems (CASES \u201912), pp. 131\u2013140. ACM.","DOI":"10.1145\/2380403.2380430"},{"key":"957_CR9","unstructured":"She, D., & et al. (2012). Scheduling for register file energy minimization in explicit datapath architectures. In: Design, Automation Test in Europe Conference Exhibition, 2012 (DATE \u201912), pp. 388\u2013393. EDAA."},{"key":"957_CR10","doi-asserted-by":"crossref","unstructured":"She, D., & et al. (2013). OpenCL Code Generation for Low Energy Wide SIMD Architectures with Explicit Datapath. In: Proceedings of the 13th International Conference on Embedded Computer Systems (SAMOS-XIII), pp. 322\u2013329. IEEE.","DOI":"10.1109\/SAMOS.2013.6621141"},{"key":"957_CR11","unstructured":"Corporaal, H. (1998). Microprocessor Architectures, From VLIW to TTA. Wiley."},{"issue":"1","key":"957_CR12","doi-asserted-by":"crossref","first-page":"17","DOI":"10.1109\/L-CA.2011.26","volume":"11","author":"I Finlayson","year":"2012","unstructured":"Finlayson, I., & et al. (2012). An overview of static pipelining. Computer Architecture Letters, 11(1), 17\u201320.","journal-title":"Computer Architecture Letters"},{"issue":"1","key":"957_CR13","doi-asserted-by":"crossref","first-page":"29","DOI":"10.1109\/L-CA.2008.1","volume":"7","author":"J Balfour","year":"2007","unstructured":"Balfour, J., & et al. (2007). An energy-efficient processor architecture for embedded systems. Computer Architecture Letters, 7(1), 29\u201332.","journal-title":"Computer Architecture Letters"},{"issue":"2","key":"957_CR14","doi-asserted-by":"crossref","first-page":"60","DOI":"10.1109\/L-CA.2009.45","volume":"8","author":"J Balfour","year":"2009","unstructured":"Balfour, J., & et al. (2009). Operand registers and explicit operand forwarding. Computer Architecture Letters, 8(2), 60\u201363.","journal-title":"Computer Architecture Letters"},{"key":"957_CR15","doi-asserted-by":"crossref","unstructured":"Heikkinen, J., & et al. (2005). Dictionary-based program compression on TTAs: effects on area and power consumption. In: Proceedings of the 2005 IEEE Workshop on Signal Processing Systems Design and Implementation, pp. 479\u2013484.","DOI":"10.1109\/SIPS.2005.1579916"},{"key":"957_CR16","unstructured":"Khronos OpenCL Working Group: The OpenCL Specification, version 1.2 (2012). URL http:\/\/www.khronos.org\/registry\/cl\/ ."},{"key":"957_CR17","doi-asserted-by":"crossref","unstructured":"Waeijen, L., & et al. (2013). SIMD Made Explicit. In: Proceedings of the 13th International Conference on Embedded Computer Systems (SAMOS-XIII), pp. 330\u2013337. IEEE.","DOI":"10.1109\/SAMOS.2013.6621142"},{"key":"957_CR18","doi-asserted-by":"crossref","unstructured":"Owaida, M., & et al. (2011). Synthesis of platform architectures from OpenCL programs. In: Proceedings of the 19th International Symposium on Field Programmable Custom Computing Machines (FCCM \u201911), pp. 186\u2013193. IEEE.","DOI":"10.1109\/FCCM.2011.19"},{"key":"957_CR19","doi-asserted-by":"crossref","unstructured":"Woh, M., & et al. (2009). AnySP: anytime anywhere anyway signal processing. In: Proceedings of the 36th Annual International Symposium on Computer Architecture (ISCA \u201909), pp. 128\u2013139.","DOI":"10.1145\/1555754.1555773"},{"key":"957_CR20","doi-asserted-by":"crossref","unstructured":"Esko, O., & et al. (2010). Customized exposed datapath soft-core design flow with compiler support. In: Proceedings of 20th International Conference on Field Programmable Logic and Applications, pp. 217\u2013222.","DOI":"10.1109\/FPL.2010.51"},{"key":"957_CR21","doi-asserted-by":"crossref","unstructured":"J\u00e4\u00e4skel\u00e4inen, P, & et al. (2010). OpenCL-based design methodology for application-specific processors. In: Proceedings of the 10th International Conference on Embedded Computer Systems (SAMOS-X), pp. 223\u2013230.","DOI":"10.1109\/ICSAMOS.2010.5642061"},{"key":"957_CR22","doi-asserted-by":"crossref","unstructured":"Govindarajan, R., & et al. (2001). Minimum Register Instruction Sequence Problem: Revisiting Optimal Code Generation for DAGs. In: Proceedings of the 15th International Parallel & Distributed Processing Symposium (IPDPS \u201901), pp. 26\u201333. IEEE Computer Society.","DOI":"10.1109\/IPDPS.2001.924962"},{"key":"957_CR23","doi-asserted-by":"crossref","unstructured":"Karrenberg, R., & Hack, S. (2012). Improving performance of OpenCL on CPUs. In: Proceedings of the 21st International Conference on Compiler Construction (CC \u201912), pp. 1\u201320. Springer-Verlag.","DOI":"10.1007\/978-3-642-28652-0_1"},{"issue":"4","key":"957_CR24","doi-asserted-by":"crossref","first-page":"715","DOI":"10.1145\/321607.321620","volume":"17","author":"R Sethi","year":"1970","unstructured":"Sethi, R., & Ullman, J. D. (1970). The generation of optimal code for arithmetic expressions. Journal of the ACM, 17(4), 715\u2013728.","journal-title":"Journal of the ACM"},{"key":"957_CR25","doi-asserted-by":"crossref","unstructured":"Park, S., & et al. (2006). Bypass aware instruction scheduling for register file power reduction. In: Proceedings of the 2006 ACM Conference on Language, Compilers, and Tool Support for Embedded Systems (LCTES \u201906), pp. 173\u2013181. ACM.","DOI":"10.1145\/1134650.1134675"},{"key":"957_CR26","doi-asserted-by":"crossref","unstructured":"Guzma, V., & et al. (2009). Reducing processor energy consumption by compiler optimization. In: IEEE Workshop on Signal Processing Systems (SiPS), pp. 63\u201368.","DOI":"10.1109\/SIPS.2009.5336226"},{"key":"957_CR27","doi-asserted-by":"crossref","unstructured":"Guzma, V., & et al. (2013). Use of compiler optimization of software bypassing as a method to improve energy efficiency of exposed data path architectures. EURASIP Journal on Embedded Systems, 2013 (1).","DOI":"10.1186\/1687-3963-2013-9"},{"key":"957_CR28","doi-asserted-by":"crossref","unstructured":"He, Y., & et al. (2010). Xetal-Pro: An Ultra-Low Energy and High Throughput SIMD Processor. In: Proceedings of the 47th Annual Design Automation Conference (DAC \u201910), pp. 543\u2013548.","DOI":"10.1145\/1837274.1837409"},{"key":"957_CR29","doi-asserted-by":"crossref","unstructured":"He, Y., & et al. (2011). MOVE-Pro: a low power and high code density tta architecture. In: Proceedings of the 11th International Conference on Embedded Computer Systems (SAMOS-XI), pp. 294\u2013301.","DOI":"10.1109\/SAMOS.2011.6045474"},{"issue":"4","key":"957_CR30","doi-asserted-by":"crossref","first-page":"472","DOI":"10.1109\/TCSVT.2011.2125590","volume":"21","author":"Y Pu","year":"2011","unstructured":"Pu, Y., & et al. (2011). From Xetal-II to Xetal-Pro: On the Road Toward an Ultra-Low-Energy and High-Throughput SIMD processor. IEEE Transactions on Circuits and Systems for Video Technology, 21(4), 472\u2013484.","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"}],"container-title":["Journal of Signal Processing Systems"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11265-014-0957-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11265-014-0957-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11265-014-0957-1","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,8,15]],"date-time":"2019-08-15T10:00:45Z","timestamp":1565863245000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11265-014-0957-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2014,9,28]]},"references-count":30,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2015,7]]}},"alternative-id":["957"],"URL":"https:\/\/doi.org\/10.1007\/s11265-014-0957-1","relation":{},"ISSN":["1939-8018","1939-8115"],"issn-type":[{"value":"1939-8018","type":"print"},{"value":"1939-8115","type":"electronic"}],"subject":[],"published":{"date-parts":[[2014,9,28]]}}}