{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T13:39:13Z","timestamp":1740145153156,"version":"3.37.3"},"reference-count":52,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2015,11,20]],"date-time":"2015-11-20T00:00:00Z","timestamp":1447977600000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"funder":[{"name":"EU project P-SOCRATES","award":["611016"],"award-info":[{"award-number":["611016"]}]},{"name":"EU project MULTITHERMAN","award":["291125"],"award-info":[{"award-number":["291125"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Real-Time Image Proc"],"published-print":{"date-parts":[[2018,6]]},"DOI":"10.1007\/s11554-015-0544-0","type":"journal-article","created":{"date-parts":[[2015,11,20]],"date-time":"2015-11-20T01:35:12Z","timestamp":1447983312000},"page":"73-92","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":11,"title":["Optimizing memory bandwidth exploitation for OpenVX applications on embedded many-core accelerators"],"prefix":"10.1007","volume":"15","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9221-4633","authenticated-orcid":false,"given":"Giuseppe","family":"Tagliavini","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Germain","family":"Haugou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Andrea","family":"Marongiu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Luca","family":"Benini","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2015,11,20]]},"reference":[{"key":"544_CR1","unstructured":"Adapteva, Inc (2015) Epiphany-IV 64-core 28nm Microprocessor. http:\/\/www.adapteva.com\/products\/silicon-devices\/e64g401\/"},{"key":"544_CR2","doi-asserted-by":"crossref","unstructured":"Agosta, G., Barenghi, A., Pelosi, G., Scandale, M.: Towards transparently tackling functionality and performance issues across different OpenCL platforms. In: 2014\u00a0Second International Symposium on Computing and Networking (CANDAR), pp. 130\u2013136. IEEE (2014)","DOI":"10.1109\/CANDAR.2014.53"},{"key":"544_CR3","doi-asserted-by":"crossref","first-page":"440","DOI":"10.1007\/s10766-010-0135-4","volume":"38","author":"E Ayguad\u00e9","year":"2010","unstructured":"Ayguad\u00e9, E., Badia, R.M., Bellens, P., Cabrera, D., Duran, A., Ferrer, R., Gonz\u00e0lez, M., Igual, F., Jim\u00e9nez-Gonz\u00e1lez, D., Labarta, J. et al.: Extending OpenMP to survive the heterogeneous multi-core era. Int. J. Parallel Program. 38, 440\u2013459 (2010)","journal-title":"Int. J. Parallel Program."},{"key":"544_CR4","doi-asserted-by":"crossref","unstructured":"Benini, L., Flamand, E., Fuin, D., Melpignano, D.: P2012: building an ecosystem for a scalable, modular and high-efficiency embedded computing accelerator. In: Design, Automation Test in Europe Conference Exhibition (DATE), pp. 983\u2013987. IEEE (2012)","DOI":"10.1109\/DATE.2012.6176639"},{"key":"544_CR5","unstructured":"Boudier, P., Sellers, G.: Memory system on fusion APUs: the benefits of zero copy. In: AMD Fusion Developer Summit. AMD (2011). http:\/\/www.developer.amd.com\/afds\/assets\/presentations\/1004_final.pdf"},{"key":"544_CR6","doi-asserted-by":"crossref","unstructured":"Canis, A., Choi, J., Aldham, M., Zhang, V., Kammoona, A., Anderson, J.H., Brown, S., Czajkowski, T.: LegUp: high-level synthesis for FPGA-based processor\/accelerator systems. In: Proceedings of the 19th ACM\/SIGDA International Symposium on Field Programmable Gate Arrays, pp.\u00a033\u201336. ACM (2011)","DOI":"10.1145\/1950413.1950423"},{"key":"544_CR7","doi-asserted-by":"crossref","unstructured":"Canny, J.: A computational approach to edge detection. In:\u00a0IEEE Transactions on Pattern Analysis and Machine Intelligence, vol. 6, pp. 679\u2013698. IEEE (1986)","DOI":"10.1109\/TPAMI.1986.4767851"},{"key":"544_CR8","doi-asserted-by":"crossref","unstructured":"Cong, J., Liu, C., Ghodrat, M.A., Reinman, G., Gill, M., Zou, Y.: AXR-CMP: architecture support in accelerator-rich CMPs. In:\u00a02nd Workshop on SoC Architecture, Accelerators and Workloads (2011)","DOI":"10.1145\/2228360.2228512"},{"key":"544_CR9","doi-asserted-by":"crossref","unstructured":"Cong, J., Ghodrat, M.A,, Gill, M., Grigorian, B., Reinman, G.: CHARM: a composable heterogeneous accelerator-rich microprocessor. In: Proceedings of the 2012 ACM\/IEEE International Symposium on Low Power Electronics and Design, pp.\u00a0379\u2013384. ACM (2012)","DOI":"10.1145\/2333660.2333747"},{"key":"544_CR10","doi-asserted-by":"crossref","unstructured":"Conti, F., Rossi, D., Pullini, A., Loi, I., Benini, L.: Energy-efficient vision on the PULP platform for ultra-low power parallel computing. In: 2014 IEEE Workshop on Signal Processing Systems (SiPS), pp. 1\u20136. IEEE (2014)","DOI":"10.1109\/SiPS.2014.6986099"},{"issue":"3","key":"544_CR11","doi-asserted-by":"crossref","first-page":"260","DOI":"10.7227\/IJEEE.49.3.6","volume":"49","author":"J Coombs","year":"2012","unstructured":"Coombs, J., Prabhu, R., Peake, G.: Overcoming the challenges of porting OpenCV to TI's embedded ARM+ DSP platforms. Int. J. Electr. Eng. Educ. 49(3), 260\u2013274 (2012)","journal-title":"Int. J. Electr. Eng. Educ."},{"key":"544_CR12","doi-asserted-by":"crossref","unstructured":"Czajkowski, T.S., Aydonat, U., Denisenko, D., Freeman, J., Kinsner, M., Neto, D., Wong, J., Yiannacouras, P., Singh, DP.: From OpenCL to high-performance hardware on FPGAs. In: 22nd International Conference on Field Programmable Logic and Applications (FPL), pp. 531\u2013534. IEEE (2012)","DOI":"10.1109\/FPL.2012.6339272"},{"issue":"1","key":"544_CR13","doi-asserted-by":"crossref","first-page":"129","DOI":"10.1137\/070693199","volume":"51","author":"K Datta","year":"2009","unstructured":"Datta, K., Kamil, S., Williams, S., Oliker, L., Shalf, J., Yelick, K.: Optimization and performance modeling of stencil computations on modern microprocessors. SIAM Rev. 51(1), 129\u2013159 (2009)","journal-title":"SIAM Rev."},{"key":"544_CR14","unstructured":"Embedded Vision Alliance (2015) Website. http:\/\/www.embedded-vision.com\/"},{"key":"544_CR15","doi-asserted-by":"crossref","unstructured":"Farabet, C., Martini, B., Corda, B., Akselrod, P., Culurciello, E., LeCun, Y.: Neuflow: a runtime reconfigurable dataflow processor for vision. In: 2011 IEEE Computer Society Conference on Computer Vision and Pattern Recognition Workshops (CVPRW), pp. 109\u2013116. IEEE (2011)","DOI":"10.1109\/CVPRW.2011.5981829"},{"key":"544_CR16","doi-asserted-by":"crossref","unstructured":"Fatahalian, K., Horn, DR., Knight, T.J., Leem, L., Houston, M., Park, J.Y., Erez , M., Ren, M., Aiken, A., Dally, W.J. et\u00a0al: Sequoia: programming the memory hierarchy. In: Proceedings of the 2006 ACM\/IEEE conference on Supercomputing, p. 83. ACM (2006)","DOI":"10.1145\/1188455.1188543"},{"key":"544_CR17","doi-asserted-by":"crossref","unstructured":"Franceschelli, A., Burgio, P., Tagliavini, G., Marongiu, A., Ruggiero, M., Lombardi, M., Bonfietti, A., Milano, M., Benini, L.: MPOpt-Cell: a high-performance data-flow programming environment for the CELL BE processor. In: Proceedings of the 8th ACM International Conference on Computing Frontiers, p. 11. ACM (2011)","DOI":"10.1145\/2016604.2016618"},{"key":"544_CR18","doi-asserted-by":"crossref","unstructured":"Gehrig, S.K., Eberli, F., Meyer, T.: A real-time low-power stereo vision engine using semi-global matching. In: Computer Vision Systems, pp.\u00a0134\u2013143. Springer (2009)","DOI":"10.1007\/978-3-642-04667-4_14"},{"key":"544_CR19","doi-asserted-by":"crossref","unstructured":"Geilen, M., Basten, T., Stuijk, S.: Minimising buffer requirements of synchronous dataflow graphs with model checking. In: Proceedings of the 42nd annual Design Automation Conference, pp.\u00a0819\u2013824. ACM (2005)","DOI":"10.1145\/1065579.1065796"},{"key":"544_CR20","doi-asserted-by":"crossref","unstructured":"Gonz\u00e0lez, M., Vujic, N., Martorell, X., Ayguad\u00e9, E., Eichenberger, A.E., Chen, T., Sura, Z., Zhang, T., O\u2019Brien, K., O\u2019Brien, K.: Hybrid access-specific software cache techniques for the Cell BE architecture. In: Proceedings of the 17th International Conference on Parallel Architectures and Compilation Techniques, pp.\u00a0292\u2013302. ACM (2008)","DOI":"10.1145\/1454115.1454156"},{"issue":"2","key":"544_CR21","doi-asserted-by":"crossref","first-page":"19","DOI":"10.1145\/2557445","volume":"57","author":"S Greengard","year":"2014","unstructured":"Greengard, S.: Computational photography comes into focus. Commun. ACM 57(2), 19\u201321 (2014)","journal-title":"Commun. ACM"},{"key":"544_CR22","doi-asserted-by":"crossref","unstructured":"Hegarty, J., Brunhaver, J., DeVito, Z., Ragan-Kelley, J., Cohen, N., Bell, S., Vasilyev, A., Horowitz, M., Hanrahan, P. Darkroom: Compiling high-level image processing code into hardware pipelines. In: Proceedings of the 41st International Conference on Computer Graphics and Interactive Techniques (SIGGRAPH) (2014)","DOI":"10.1145\/2601097.2601174"},{"issue":"2","key":"544_CR23","doi-asserted-by":"crossref","first-page":"78","DOI":"10.1109\/MCSE.2012.23","volume":"14","author":"A Heinecke","year":"2012","unstructured":"Heinecke, A., Klemm, M., Bungartz, H.: From GPGPU to many-core: Nvidia fermi and intel many integrated core architecture. Comput. Sci. Eng. 14(2), 78\u201383 (2012)","journal-title":"Comput. Sci. Eng."},{"key":"544_CR24","unstructured":"HSA Foundation Specification Library (2015).\u00a0 http:\/\/www.hsafoundation.com\/html\/HSA_Library.htm"},{"key":"544_CR25","unstructured":"KALRAY Corporation (2015) Website. http:\/\/www.kalray.eu\/"},{"key":"544_CR26","unstructured":"Kronos Group (2015a) The OpenCL 1.1 Specifications. http:\/\/www.khronos.org\/registry\/cl\/specs\/opencl-1.1.pdf"},{"key":"544_CR27","unstructured":"Kronos Group (2015b) The OpenVX API for hardware acceleration. http:\/\/www.khronos.org\/openvx"},{"key":"544_CR28","doi-asserted-by":"crossref","first-page":"42","DOI":"10.1109\/MM.2011.68","volume":"5","author":"H Lee","year":"2011","unstructured":"Lee, H., Brown, K.J., Sujeeth, A.K., Chafi, H., Rompf, T., Odersky, M., Olukotun, K.: Implementing domain-specific languages for heterogeneous parallel computing. IEEE Micro 5, 42\u201353 (2011)","journal-title":"IEEE Micro"},{"key":"544_CR29","doi-asserted-by":"crossref","unstructured":"Lee, J., Seo, S., Kim, C., Kim, J., Chun, P., Sura, Z., Kim, J., Han, S.: COMIC: a coherent shared memory interface for Cell BE. In: Proceedings of the 17th International Conference on Parallel Architectures and Compilation Techniques, pp.\u00a0303\u2013314. ACM (2008)","DOI":"10.1145\/1454115.1454157"},{"key":"544_CR30","doi-asserted-by":"crossref","unstructured":"Lei, Y., Gang, Z., Si-Heon, R., Choon-Young, L., Sang-Ryong, L., Bae, K.M.: The platform of image acquisition and processing system based on DSP and FPGA. In: International Conference on Smart Manufacturing Application, pp.\u00a0470\u2013473. IEEE (2008)","DOI":"10.1109\/ICSMA.2008.4505567"},{"key":"544_CR31","doi-asserted-by":"crossref","unstructured":"Lepley, T., Paulin, P., Flamand, E. A novel compilation approach for image processing graphs on a many-core platform with explicitly managed memory. In: Proceedings of the 2013 International Conference on Compilers, Architectures and Synthesis for Embedded Systems, pp.\u00a01\u201310. IEEE (2013)","DOI":"10.1109\/CASES.2013.6662510"},{"key":"544_CR32","unstructured":"Lucas, B.D., Kanade, T.: An iterative image registration technique with an application to stereo vision. In: IJCAI,\u00a0vol. 81, pp. 674\u2013679. IJCAI Organization (1981)"},{"key":"544_CR33","doi-asserted-by":"crossref","unstructured":"Maghazeh, A., Bordoloi, U.D., Eles, P., Peng, Z.: General purpose computing on low-power embedded GPUs: has it come of age? In: 2013 International Conference on Embedded Computer Systems: Architectures, Modeling, and Simulation (SAMOS XIII), pp. 1\u201310. IEEE (2013)","DOI":"10.1109\/SAMOS.2013.6621099"},{"key":"544_CR34","doi-asserted-by":"crossref","unstructured":"Magno, M., Tombari, F., Brunelli, D., Di Stefano, L., Benini, L.: Multimodal abandoned\/removed object detection for low power video surveillance systems. In: Sixth IEEE International Conference on Advanced Video and Signal Based Surveillance, pp.\u00a0188\u2013193. IEEE (2009)","DOI":"10.1109\/AVSS.2009.72"},{"key":"544_CR35","doi-asserted-by":"publisher","unstructured":"Membarth, R., Reiche, O., Hannig, F., Teich, J., Korner, M., Eckert, W.: HIPAcc: a Domain-Specific Language and Compiler for Image Processing. IEEE Trans. Parallel Distrib. Syst. doi: 10.1109\/TPDS.2015.2394802 (2015)","DOI":"10.1109\/TPDS.2015.2394802"},{"key":"544_CR36","unstructured":"Movidius, L.D.T.: Myriad 1 Mobile Vision Processor. http:\/\/www.movidius.com\/our-technology\/myriad-2-platform\/ (2015)"},{"key":"544_CR37","unstructured":"NVIDIA (2015) Tegra Android Development Documentation Website. http:\/\/docs.nvidia.com\/tegra\/index.html"},{"key":"544_CR38","unstructured":"OpenCV Library Homepage (2015) Website. http:\/\/www.opencv.com\/"},{"key":"544_CR39","doi-asserted-by":"crossref","unstructured":"Park, S., Maashri, A.A., Irick, K.M., Chandrashekhar, A., Cotter, M., Chandramoorthy, N., Debole, M., Narayanan, V.: System-on-chip for biologically inspired vision applications. IPSJ Trans. Syst. LSI Design Methodol. 5, 71\u201395 (2012)","DOI":"10.2197\/ipsjtsldm.5.71"},{"key":"544_CR40","unstructured":"Plurality Ltd (2015) The HyperCore Processor. http:\/\/www.plurality.com\/hypercore.html"},{"key":"544_CR41","unstructured":"Qualcomm (2015) Computer Vision (FastCV). https:\/\/developer.qualcomm.com\/computer-vision-fastcv"},{"key":"544_CR42","doi-asserted-by":"crossref","unstructured":"Ragan-Kelley, J., Barnes, C., Adams, A., Paris, S., Durand, F., Amarasinghe, S.: Halide: a language and compiler for optimizing parallelism, locality, and recomputation in image processing pipelines. In: Proceedings of the 34th ACM SIGPLAN Conference on Programming Language Design and Implementation,\u00a0vol. 48, pp. 519\u2013530. ACM (2013)","DOI":"10.1145\/2491956.2462176"},{"key":"544_CR43","doi-asserted-by":"crossref","unstructured":"Rainey, E., Villarreal, J., Dedeoglu, G., Pulli, K., Lepley, T., Brill, F. Addressing System-Level Optimization with OpenVX Graphs. In: 2014 IEEE Conference on Computer Vision and Pattern Recognition Workshops (CVPRW), pp.\u00a0658\u2013663. IEEE (2014)","DOI":"10.1109\/CVPRW.2014.100"},{"issue":"1","key":"544_CR44","doi-asserted-by":"crossref","first-page":"105","DOI":"10.1109\/TPAMI.2008.275","volume":"32","author":"E Rosten","year":"2010","unstructured":"Rosten, E., Porter, R., Drummond, T.: Faster and better: a machine learning approach to corner detection. IEEE Trans. Patter. Anal. Mach. Intell. 32(1), 105\u2013119 (2010)","journal-title":"IEEE Trans. Patter. Anal. Mach. Intell."},{"key":"544_CR45","doi-asserted-by":"crossref","unstructured":"Schubert, F., Schertler, K., Mikolajczyk, K.: A hands-on approach to high-dynamic-range and super resolution fusion. In: 2009 Workshop on\u00a0Applications of Computer Vision (WACV), pp. 1\u20138. IEEE (2009)","DOI":"10.1109\/WACV.2009.5403080"},{"key":"544_CR46","unstructured":"Sonka, M., Hlavac, V., Boyle, R..: Image processing, analysis, and machine vision. Thomson Toronto (2008)"},{"key":"544_CR47","doi-asserted-by":"crossref","first-page":"66","DOI":"10.1109\/MCSE.2010.69","volume":"12","author":"JE Stone","year":"2010","unstructured":"Stone, J.E., Gohara, D., Shi, G.: OpenCL: a parallel programming standard for heterogeneous computing systems. Comput. Sci. Eng. 12, 66\u201373 (2010)","journal-title":"Comput. Sci. Eng."},{"key":"544_CR48","doi-asserted-by":"crossref","unstructured":"Tagliavini, G., Haugou, G., Marongiu, A., Benini, L.: A framework for optimizing OpenVX applications performance on embedded manycore accelerators. In: Proceedings of the 18th International Workshop on Software and Compilers for Embedded Systems, pp. 125\u2013128. ACM (2015)","DOI":"10.1145\/2764967.2776858"},{"key":"544_CR49","doi-asserted-by":"crossref","unstructured":"Thies, W., Karczmarek, M., Amarasinghe, S.: StreamIt: a language for streaming applications. In: Compiler Construction, pp.\u00a0179\u2013196. Springer (2002)","DOI":"10.1007\/3-540-45937-5_14"},{"key":"544_CR50","doi-asserted-by":"crossref","unstructured":"Vajda, A.: Programming many-core chips. Springer (2011)","DOI":"10.1007\/978-1-4419-9739-5"},{"key":"544_CR51","doi-asserted-by":"crossref","unstructured":"Wienke, S., Springer, P., Terboven, C., an\u00a0Mey, D. OpenACC First Experiences with Real-World Applications. In: Euro-Par 2012 Parallel Processing, pp.\u00a0859\u2013870. Springer (2012)","DOI":"10.1007\/978-3-642-32820-6_85"},{"key":"544_CR52","unstructured":"Zedboard.org (2015) Zedboard product page. http:\/\/zedboard.org\/product\/zedboard"}],"container-title":["Journal of Real-Time Image Processing"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11554-015-0544-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11554-015-0544-0\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11554-015-0544-0","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11554-015-0544-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,9,1]],"date-time":"2019-09-01T14:37:37Z","timestamp":1567348657000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11554-015-0544-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2015,11,20]]},"references-count":52,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2018,6]]}},"alternative-id":["544"],"URL":"https:\/\/doi.org\/10.1007\/s11554-015-0544-0","relation":{},"ISSN":["1861-8200","1861-8219"],"issn-type":[{"type":"print","value":"1861-8200"},{"type":"electronic","value":"1861-8219"}],"subject":[],"published":{"date-parts":[[2015,11,20]]}}}