{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,27]],"date-time":"2026-03-27T23:23:53Z","timestamp":1774653833702,"version":"3.50.1"},"reference-count":37,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2022,8,16]],"date-time":"2022-08-16T00:00:00Z","timestamp":1660608000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,8,16]],"date-time":"2022-08-16T00:00:00Z","timestamp":1660608000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"published-print":{"date-parts":[[2023,2]]},"DOI":"10.1007\/s11227-022-04749-0","type":"journal-article","created":{"date-parts":[[2022,8,16]],"date-time":"2022-08-16T16:03:08Z","timestamp":1660665788000},"page":"2404-2430","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":8,"title":["Godiva: green on-chip interconnection for DNNs"],"prefix":"10.1007","volume":"79","author":[{"given":"Arghavan","family":"Asad","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Farah","family":"Mohammadi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,8,16]]},"reference":[{"key":"4749_CR1","unstructured":"Inci A, Bolotin E, Fu Y, Dalal G, Mannor S, Nellans D, Marculescu D (2020) The architectural implications of distributed reinforcement learning on CPU-GPU systems. arXiv:2012.04210"},{"issue":"3","key":"4749_CR2","doi-asserted-by":"publisher","first-page":"211","DOI":"10.1007\/s11263-015-0816-y","volume":"115","author":"Olga Russakovsky","year":"2015","unstructured":"Russakovsky Olga, Deng Jia, Hao Su, Krause Jonathan, Satheesh Sanjeev, Ma Sean, Huang Zhiheng et al (2015) Imagenet large scale visual recognition challenge. Int J Comput Vis 115(3):211\u2013252","journal-title":"Int J Comput Vis"},{"key":"4749_CR3","doi-asserted-by":"crossref","unstructured":"Bakhoda A, Yuan GL, Fung WW, Wong H, Aamodt TM (2009) Analyzing CUDA workloads using a detailed GPU simulator. In: 2009 IEEE International Symposium on Performance Analysis of Systems and Software, pp 163\u2013174. IEEE","DOI":"10.1109\/ISPASS.2009.4919648"},{"key":"4749_CR4","unstructured":"Espeholt L, Marinier R, Stanczyk P, Wang K, Michalski M (2019) Seed rl: scalable and efficient deep-rl with accelerated central inference. arXiv:1910.06591"},{"key":"4749_CR5","doi-asserted-by":"crossref","unstructured":"Kayiran O, Nachiappan NC, Jog A, Ausavarungnirun R, Kandemir MT, Loh GH, Mutlu O, Das CR (2014) Managing GPU concurrency in heterogeneous architectures. In: 2014 47th Annual IEEE\/ACM International Symposium on Microarchitecture, pp 114\u2013126. IEEE","DOI":"10.1109\/MICRO.2014.62"},{"issue":"7","key":"4749_CR6","doi-asserted-by":"publisher","first-page":"66","DOI":"10.1109\/MC.2018.3011040","volume":"51","author":"Ryan Gary Kim","year":"2018","unstructured":"Kim Ryan Gary, Doppa Janardhan Rao, Pande Partha Pratim, Marculescu Diana, Marculescu Radu (2018) Machine learning and manycore systems design: a serendipitous symbiosis. Computer 51(7):66\u201377","journal-title":"Computer"},{"issue":"12","key":"4749_CR7","doi-asserted-by":"publisher","first-page":"2295","DOI":"10.1109\/JPROC.2017.2761740","volume":"105","author":"V Sze","year":"2017","unstructured":"Sze V, Chen Y-H, Yang T-J, Emer JS (2017) Efficient processing of deep neural networks: a tutorial and survey. Proc IEEE 105(12):2295\u20132329","journal-title":"Proc IEEE"},{"key":"4749_CR8","unstructured":"Inci A, Isgenc MM, Marculescu D (2020) DeepNVM++: cross-layer modeling and optimization framework of non-volatile memories for deep learning. arXiv:2012.04559"},{"issue":"3","key":"4749_CR9","doi-asserted-by":"publisher","first-page":"268","DOI":"10.1109\/JETCAS.2020.3022920","volume":"10","author":"Seyed Morteza Nabavinejad","year":"2020","unstructured":"Nabavinejad Seyed Morteza, Baharloo Mohammad, Chen Kun-Chih, Palesi Maurizio, Kogel Tim, Ebrahimi Masoumeh (2020) An overview of efficient interconnection networks for deep neural network accelerators. IEEE J Emerg Sel Top Circuits Syst 10(3):268\u2013282","journal-title":"IEEE J Emerg Sel Top Circuits Syst"},{"issue":"1","key":"4749_CR10","doi-asserted-by":"publisher","first-page":"127","DOI":"10.1109\/JSSC.2016.2616357","volume":"52","author":"Y-H Chen","year":"2016","unstructured":"Chen Y-H, Krishna T, Emer JS, Sze V (2016) Eyeriss: an energy-efficient reconfigurable accelerator for deep convolutional neural networks. IEEE J Solid-State Circuits 52(1):127\u2013138","journal-title":"IEEE J Solid-State Circuits"},{"issue":"12","key":"4749_CR11","first-page":"2198","volume":"70","author":"M Talebi","year":"2020","unstructured":"Talebi M, Salahvarzi A, Monazzah AMH, Skadron K, Fazeli M (2020) ROCKY: a robust hybrid on-chip memory kit for the processors with STT-MRAM cache technology. IEEE Trans Comput 70(12):2198\u20132210","journal-title":"IEEE Trans Comput"},{"issue":"3","key":"4749_CR12","doi-asserted-by":"publisher","first-page":"12","DOI":"10.1109\/MM.2017.54","volume":"37","author":"Y-H Chen","year":"2017","unstructured":"Chen Y-H, Emer J, Sze V (2017) Using dataflow to optimize energy efficiency of deep neural network accelerators. IEEE Micro 37(3):12\u201321","journal-title":"IEEE Micro"},{"key":"4749_CR13","doi-asserted-by":"crossref","unstructured":"Reza, MF, Ampadu P (2019) Energy-efficient and high-performance NoC architecture and mapping solution for deep neural networks. In: Proceedings of the 13th IEEE\/ACM International Symposium on Networks-on-Chip, pp 1\u20138","DOI":"10.1145\/3313231.3352377"},{"key":"4749_CR14","doi-asserted-by":"crossref","unstructured":"Mirmahaleh SYH, Reshadi M, Shabani H, Guo X, Bagherzadeh N (2019) Flow mapping and data distribution on mesh-based deep learning accelerator. In: Proceedings of the 13th IEEE\/ACM International Symposium on Networks-on-Chip, pp 1\u20138","DOI":"10.1145\/3313231.3352378"},{"issue":"1","key":"4749_CR15","doi-asserted-by":"publisher","first-page":"73","DOI":"10.1109\/TC.2016.2574353","volume":"66","author":"T Luo","year":"2016","unstructured":"Luo T, Liu S, Li L, Wang Y, Zhang S, Chen T, Zhiwei Xu, Temam O, Chen Y (2016) DaDianNao: a neural network supercomputer. IEEE Trans Comput 66(1):73\u201388","journal-title":"IEEE Trans Comput"},{"key":"4749_CR16","doi-asserted-by":"crossref","unstructured":"Liu X, Wen W, Qian X, Li H, Chen Y (2018) Neu-NoC: a high-efficient interconnection network for accelerated neuromorphic systems. In: 2018 23rd Asia and South Pacific Design Automation Conference (ASP-DAC), pp 141\u2013146. IEEE","DOI":"10.1109\/ASPDAC.2018.8297296"},{"key":"4749_CR17","doi-asserted-by":"crossref","unstructured":"Wong HSP, Raoux S, Kim S, Liang J, Reifenberg JP, Rajendran B, Asheghi M, Goodson KE (2010) Phase change memory. In: Proceedings of the IEEE 98, 12: 2201\u20132227","DOI":"10.1109\/JPROC.2010.2070050"},{"key":"4749_CR18","doi-asserted-by":"crossref","unstructured":"Liu X, Mao M, Liu B, Li B, Wang Y, Jiang H, Barnell M et al (2016) Harmonica: a framework of heterogeneous computing systems with memristor-based neuromorphic computing accelerators. In: IEEE Transactions on Circuits and Systems I: Regular Papers 63, 5: 617\u2013628","DOI":"10.1109\/TCSI.2016.2529279"},{"key":"4749_CR19","doi-asserted-by":"crossref","unstructured":"Endoh T (2021) 3D integration of memories including heterogeneous integration. In: 2021 International Symposium on VLSI Technology, Systems and Applications (VLSI-TSA), pp 1\u20132. IEEE","DOI":"10.1109\/VLSI-TSA51926.2021.9440129"},{"key":"4749_CR20","doi-asserted-by":"crossref","unstructured":"Joardar BK, Doppa JR, Pande PP, Marculescu D, Marculescu R (2018) Hybrid on-chip communication architectures for heterogeneous manycore systems. In: 2018 IEEE\/ACM International Conference on Computer-Aided Design (ICCAD), pp 1\u20136. IEEE","DOI":"10.1145\/3240765.3243480"},{"issue":"1","key":"4749_CR21","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1038\/s41598-021-82543-3","volume":"11","author":"L Bernstein","year":"2021","unstructured":"Bernstein L, Sludds A, Hamerly R, Sze V, Emer J, Englund D (2021) Freely scalable and reconfigurable optical hardware for deep learning. Sci Rep 11(1):1\u201312","journal-title":"Sci Rep"},{"issue":"1","key":"4749_CR22","doi-asserted-by":"publisher","first-page":"58","DOI":"10.1109\/MCAS.2015.2510199","volume":"16","author":"A Karkar","year":"2016","unstructured":"Karkar A, Mak T, Tong K-F, Yakovlev A (2016) A survey of emerging interconnects for on-chip efficient multicast and broadcast in many-cores. IEEE Circuits Syst Mag 16(1):58\u201372","journal-title":"IEEE Circuits Syst Mag"},{"key":"4749_CR23","first-page":"1097","volume":"25","author":"A Krizhevsky","year":"2012","unstructured":"Krizhevsky A, Sutskever I, Hinton GE (2012) Imagenet classification with deep convolutional neural networks. Adv Neural Inf Process Syst 25:1097\u20131105","journal-title":"Adv Neural Inf Process Syst"},{"key":"4749_CR24","doi-asserted-by":"crossref","unstructured":"Chatfield K, Simonyan K, Vedaldi A, Zisserman A (2014) Return of the devil in the details: delving deep into convolutional nets. arXiv:1405.3531","DOI":"10.5244\/C.28.6"},{"issue":"3","key":"4749_CR25","doi-asserted-by":"publisher","first-page":"380","DOI":"10.1145\/3007787.3001178","volume":"44","author":"D Kim","year":"2016","unstructured":"Kim D, Kung J, Chai S, Yalamanchili S, Mukhopadhyay S (2016) Neurocube: a programmable digital neuromorphic architecture with high-density 3D memory. ACM SIGARCH Comput Archit News 44(3):380\u2013392","journal-title":"ACM SIGARCH Comput Archit News"},{"issue":"1","key":"4749_CR26","doi-asserted-by":"publisher","first-page":"34","DOI":"10.1109\/LCA.2014.2299539","volume":"14","author":"J Power","year":"2014","unstructured":"Power J, Hestness J, Orr MS, Hill MD, Wood DA (2014) gem5-gpu: A heterogeneous cpu-gpu simulator. IEEE Comput Archit Lett 14(1):34\u201336","journal-title":"IEEE Comput Archit Lett"},{"issue":"2","key":"4749_CR27","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/2024716.2024718","volume":"39","author":"Nathan Binkert","year":"2011","unstructured":"Binkert Nathan, Beckmann Bradford, Black Gabriel, Reinhardt Steven K, Saidi Ali, Basu Arkaprava, Hestness Joel et al (2011) The gem5 simulator. ACM SIGARCH Comput Archit News 39(2):1\u20137","journal-title":"ACM SIGARCH Comput Archit News"},{"issue":"3","key":"4749_CR28","doi-asserted-by":"publisher","first-page":"487","DOI":"10.1145\/2508148.2485964","volume":"41","author":"Jingwen Leng","year":"2013","unstructured":"Leng Jingwen, Hetherington Tayler, ElTantawy Ahmed, Gilani Syed, Kim Nam Sung, Aamodt Tor M, Reddi Vijay Janapa (2013) GPUWattch: enabling energy optimizations in GPGPUs. ACM SIGARCH Comput Archit News 41(3):487\u2013498","journal-title":"ACM SIGARCH Comput Archit News"},{"key":"4749_CR29","doi-asserted-by":"crossref","unstructured":"Li S, Ahn JH, Strong RD, Brockman JB, Tullsen DM, Jouppi NP (2009) McPAT: an integrated power, area, and timing modeling framework for multicore and manycore architectures. In: Proceedings of the 42nd Annual IEEE\/ACM International Symposium on Microarchitecture, pp 469\u2013480","DOI":"10.1145\/1669112.1669172"},{"key":"4749_CR30","doi-asserted-by":"crossref","unstructured":"Agarwal N, Krishna T, Peh LS, Jha NK (2009) GARNET: a detailed on-chip network model inside a full-system simulator. In: 2009 IEEE International Symposium on Performance Analysis of Systems and Software, pp 33\u201342. IEEE","DOI":"10.1109\/ISPASS.2009.4919636"},{"issue":"6","key":"4749_CR31","doi-asserted-by":"publisher","first-page":"141","DOI":"10.1109\/MSP.2012.2211477","volume":"29","author":"Li Deng","year":"2012","unstructured":"Deng Li (2012) The mnist database of handwritten digit images for machine learning research [best of the web]. IEEE Signal Process Mag 29(6):141\u2013142","journal-title":"IEEE Signal Process Mag"},{"key":"4749_CR32","unstructured":"Abadi M, Agarwal A, Barham P, Brevdo E, Chen Z, Citro C, Greg S, Corrado et al (2016) Tensorflow: large-scale machine learning on heterogeneous distributed systems. arXiv:1603.04467"},{"key":"4749_CR33","doi-asserted-by":"crossref","unstructured":"Lotfi-Kamran P, Grot B, Falsafi B (2012) NOC-Out: microarchitecting a scale-out processor. In: 2012 45th Annual IEEE\/ACM International Symposium on Microarchitecture, pp 177\u2013187. IEEE","DOI":"10.1109\/MICRO.2012.25"},{"issue":"12","key":"4749_CR34","doi-asserted-by":"publisher","first-page":"1525","DOI":"10.1016\/j.jpdc.2013.07.014","volume":"73","author":"J Lee","year":"2013","unstructured":"Lee J, Li Si, Kim H, Yalamanchili S (2013) Design space exploration of on-chip ring interconnection for a CPU\u2013GPU heterogeneous architecture. J Parallel Distrib Comput 73(12):1525\u20131538","journal-title":"J Parallel Distrib Comput"},{"key":"4749_CR35","doi-asserted-by":"crossref","unstructured":"Alhubail L, Jasemi M, Bagherzadeh N (2020) Noc design methodologies for heterogeneous architecture In: 2020 28th Euromicro International Conference on Parallel, Distributed and Network-Based Processing (PDP), pp 299\u2013306. IEEE","DOI":"10.1109\/PDP50117.2020.00052"},{"issue":"3","key":"4749_CR36","doi-asserted-by":"publisher","first-page":"389","DOI":"10.1145\/2024723.2000111","volume":"39","author":"AK Mishra","year":"2011","unstructured":"Mishra AK, Vijaykrishnan N, Das CR (2011) A case for heterogeneous on-chip interconnects for CMPs. ACM SIGARCH Comput Archit News 39(3):389\u2013400","journal-title":"ACM SIGARCH Comput Archit News"},{"key":"4749_CR37","doi-asserted-by":"crossref","unstructured":"Che S, Boyer M, Meng J, Tarjan D, Sheaffer JW, Lee SH, Skadron K (2009) Rodinia: a benchmark suite for heterogeneous computing. In: 2009 IEEE International Symposium on Workload Characterization (IISWC), pp 44\u201354. IEEE","DOI":"10.1109\/IISWC.2009.5306797"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-022-04749-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11227-022-04749-0\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-022-04749-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,1,13]],"date-time":"2023-01-13T12:14:46Z","timestamp":1673612086000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11227-022-04749-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,8,16]]},"references-count":37,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2023,2]]}},"alternative-id":["4749"],"URL":"https:\/\/doi.org\/10.1007\/s11227-022-04749-0","relation":{},"ISSN":["0920-8542","1573-0484"],"issn-type":[{"value":"0920-8542","type":"print"},{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,8,16]]},"assertion":[{"value":"30 July 2022","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"16 August 2022","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no relevant financial or non-financial interests to disclose.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}