{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,8]],"date-time":"2025-10-08T16:34:07Z","timestamp":1759941247578,"version":"3.37.3"},"reference-count":45,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2024,10,22]],"date-time":"2024-10-22T00:00:00Z","timestamp":1729555200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,22]],"date-time":"2024-10-22T00:00:00Z","timestamp":1729555200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100004837","name":"Ministerio de Ciencia e Innovaci\u00f3n","doi-asserted-by":"publisher","award":["TED2021-130123B-I00","TED2021-130123B-I00","PID2021-126576NB-I00","PID2021-126576NB-I00"],"award-info":[{"award-number":["TED2021-130123B-I00","TED2021-130123B-I00","PID2021-126576NB-I00","PID2021-126576NB-I00"]}],"id":[{"id":"10.13039\/501100004837","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"published-print":{"date-parts":[[2025,1]]},"DOI":"10.1007\/s11227-024-06605-9","type":"journal-article","created":{"date-parts":[[2024,10,22]],"date-time":"2024-10-22T18:02:57Z","timestamp":1729620177000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Balanced segmentation of CNNs for multi-TPU inference"],"prefix":"10.1007","volume":"81","author":[{"given":"Jorge","family":"Villarrubia","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Luis","family":"Costero","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Francisco D.","family":"Igual","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Katzalin","family":"Olcoz","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,10,22]]},"reference":[{"key":"6605_CR1","doi-asserted-by":"publisher","first-page":"3660","DOI":"10.1109\/ACCESS.2020.3047960","volume":"9","author":"F Alshehri","year":"2021","unstructured":"Alshehri F, Muhammad G (2021) A comprehensive survey of the Internet of Things (IoT) and AI-based smart healthcare. IEEE Access 9:3660\u20133678. https:\/\/doi.org\/10.1109\/ACCESS.2020.3047960","journal-title":"IEEE Access"},{"doi-asserted-by":"publisher","unstructured":"Antonini M, Vu TH, Min C, et\u00a0al (2019) Resource characterisation of personal-scale sensing models on edge accelerators. In: Int. Workshop on Challenges in Artificial Intelligence and Machine Learning for Internet of Things, p 49-55, https:\/\/doi.org\/10.1145\/3363347.3363363","key":"6605_CR2","DOI":"10.1145\/3363347.3363363"},{"unstructured":"ASUS (2023) ASUS CRL-G18U-P3DF Datasheet. URL https:\/\/dlcdnets.asus.com\/pub\/ASUS\/mb\/AIOT\/AI_Accelerator\/AI_Accelerator_Card_Spec_Sheet.pdf?model=CRL-G18U-P3DF","key":"6605_CR3"},{"issue":"5","key":"6605_CR4","doi-asserted-by":"publisher","first-page":"16","DOI":"10.1109\/MSPEC.2019.8701189","volume":"56","author":"S Cass","year":"2019","unstructured":"Cass S (2019) Taking AI to the edge: google\u2019s TPU now comes in a maker-friendly package. IEEE Spectrum 56(5):16\u201317. https:\/\/doi.org\/10.1109\/MSPEC.2019.8701189","journal-title":"IEEE Spectrum"},{"issue":"7","key":"6605_CR5","doi-asserted-by":"publisher","first-page":"1665","DOI":"10.1109\/TPDS.2020.3041474","volume":"32","author":"J Du","year":"2021","unstructured":"Du J, Zhu X, Shen M et al (2021) Model parallelism optimization for distributed inference via decoupled cnn structure. IEEE Trans on Paral and Distrib Syst 32(7):1665\u20131676. https:\/\/doi.org\/10.1109\/TPDS.2020.3041474","journal-title":"IEEE Trans on Paral and Distrib Syst"},{"unstructured":"Google Coral (2019) M.2 Accelerator A+E key datasheet. URL https:\/\/coral.ai\/docs\/m2\/datasheet\/","key":"6605_CR6"},{"doi-asserted-by":"publisher","unstructured":"Guo J, Liu W, Wang W, et\u00a0al (2019) Accudnn: A gpu memory efficient accelerator for training ultra-deep neural networks. In: 2019 IEEE 37th International Conference on Computer Design (ICCD), pp 65\u201372, https:\/\/doi.org\/10.1109\/ICCD46524.2019.00017","key":"6605_CR7","DOI":"10.1109\/ICCD46524.2019.00017"},{"issue":"10","key":"6605_CR8","doi-asserted-by":"publisher","first-page":"9241","DOI":"10.1109\/JIOT.2020.2981338","volume":"7","author":"W He","year":"2020","unstructured":"He W, Guo S, Guo S et al (2020) Joint dnn partition deployment and resource allocation for delay-sensitive deep learning inference in iot. IEEE Internet of Things J 7(10):9241\u20139254. https:\/\/doi.org\/10.1109\/JIOT.2020.2981338","journal-title":"IEEE Internet of Things J"},{"doi-asserted-by":"publisher","unstructured":"Hu C, Bao W, Wang D, et\u00a0al (2019) Dynamic adaptive dnn surgery for inference acceleration on the edge. In: IEEE INFOCOM 2019-IEEE Conference on Computer Communications, IEEE, pp 1423\u20131431, https:\/\/doi.org\/10.1109\/INFOCOM.2019.8737614","key":"6605_CR9","DOI":"10.1109\/INFOCOM.2019.8737614"},{"doi-asserted-by":"publisher","unstructured":"Huang Y, Cheng Y, Bapna A, et\u00a0al (2019) Gpipe: Efficient training of giant neural networks using pipeline parallelism. In: Wallach H, Larochelle H, Beygelzimer A, et\u00a0al (eds) Advances in Neural Information Processing Systems, vol\u00a032. Curran Associates, Inc., https:\/\/doi.org\/10.48550\/arXiv.1811.06965","key":"6605_CR10","DOI":"10.48550\/arXiv.1811.06965"},{"unstructured":"Hunmin Yang, Se-Yoon Oh, Ki-Jung Ryu (2019) Accelerating distributed deep learning inference on multi-GPU with Hadoop Spark. URL https:\/\/developer.download.nvidia.com\/video\/gputechconf\/gtc\/2019\/presentation\/s9343-accelerating-distributed-deep-learning-inference-on-multi-gpu-with-hadoop-spark_V2.pdf","key":"6605_CR11"},{"unstructured":"Intel Corp. (2017) Intel Movidius Myriad X Vision Processing Unit (VPU) with Neural Compute Engine. URL https:\/\/www.intel.com\/content\/www\/us\/en\/products\/docs\/processors\/movidius-vpu\/myriad-x-product-brief.html","key":"6605_CR12"},{"issue":"2","key":"6605_CR13","doi-asserted-by":"publisher","first-page":"139","DOI":"10.1109\/LCA.2020.3023723","volume":"19","author":"A Jahanshahi","year":"2020","unstructured":"Jahanshahi A, Sabzi HZ, Lau C et al (2020) Gpu-nest: characterizing energy efficiency of multi-gpu inference servers. IEEE Comput Architec Lett 19(2):139\u2013142. https:\/\/doi.org\/10.1109\/LCA.2020.3023723","journal-title":"IEEE Comput Architec Lett"},{"doi-asserted-by":"publisher","unstructured":"James A, Sirakoulis GC, Roy K (2019) Smart cameras everywhere: AI vision on edge with emerging memories. In: IEEE Int. Conf. on Electronics, Circuits and Systems (ICECS), pp 422\u2013425, https:\/\/doi.org\/10.1109\/ICECS46596.2019.8965029","key":"6605_CR14","DOI":"10.1109\/ICECS46596.2019.8965029"},{"doi-asserted-by":"publisher","unstructured":"Jouppi NP, Young C, Patil N, et\u00a0al (2017) In-Datacenter Performance Analysis of a Tensor Processing Unit. In: Int. Sym. on Computer Architecture, p 1-12, https:\/\/doi.org\/10.1145\/3079856.3080246","key":"6605_CR15","DOI":"10.1145\/3079856.3080246"},{"issue":"7","key":"6605_CR16","doi-asserted-by":"publisher","first-page":"67","DOI":"10.1145\/3360307","volume":"63","author":"NP Jouppi","year":"2020","unstructured":"Jouppi NP, Yoon DH, Kurian G et al (2020) A domain-specific supercomputer for training deep neural networks. Commun ACM 63(7):67\u201378. https:\/\/doi.org\/10.1145\/3360307","journal-title":"Commun ACM"},{"key":"6605_CR17","doi-asserted-by":"publisher","DOI":"10.3390\/math10224299","author":"P Kang","year":"2022","unstructured":"Kang P, Somtham A (2022) An evaluation of modern accelerator-based edge devices for object detection applications. Mathematics. https:\/\/doi.org\/10.3390\/math10224299","journal-title":"Mathematics"},{"unstructured":"Kim J, Lee JH, Kim S, et\u00a0al (2023) Memory-efficient fine-tuning of compressed large language models via sub-4-bit integer quantization. In: Oh A, Naumann T, Globerson A, et\u00a0al (eds) Advances in Neural Information Processing Systems, vol\u00a036. Curran Associates, Inc., pp 36187\u201336207, URL https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2023\/file\/7183f4fc87598f6c6e947b96714acbd6-Paper-Conference.pdf","key":"6605_CR18"},{"doi-asserted-by":"publisher","unstructured":"Kumar S, Bradbury J, Young C, et\u00a0al (2021) Exploring the limits of concurrency in ml training on google tpus. https:\/\/doi.org\/10.48550\/arXiv.2011.03641","key":"6605_CR19","DOI":"10.48550\/arXiv.2011.03641"},{"issue":"1","key":"6605_CR20","doi-asserted-by":"publisher","first-page":"447","DOI":"10.1109\/TWC.2019.2946140","volume":"19","author":"E Li","year":"2020","unstructured":"Li E, Zeng L, Zhou Z et al (2020) Edge AI: on-demand accelerating deep neural network inference via edge computing. IEEE Trans on Wire Commun 19(1):447\u2013457. https:\/\/doi.org\/10.1109\/TWC.2019.2946140","journal-title":"IEEE Trans on Wire Commun"},{"issue":"5","key":"6605_CR21","doi-asserted-by":"publisher","first-page":"3017","DOI":"10.1109\/TMC.2021.3125949","volume":"22","author":"J Li","year":"2021","unstructured":"Li J, Liang W, Li Y et al (2021) Throughput maximization of delay-aware dnn inference in edge computing by exploring dnn model partitioning and inference parallelism. IEEE Trans Mobile Comput 22(5):3017\u20133030. https:\/\/doi.org\/10.1109\/TMC.2021.3125949","journal-title":"IEEE Trans Mobile Comput"},{"unstructured":"Li Z, Zheng L, Zhong Y, et\u00a0al (2023) AlpaServe: Statistical multiplexing with model parallelism for deep learning serving. In: 17th USENIX Symposium on Operating Systems Design and Implementation (OSDI 23). USENIX Association, Boston, MA, pp 663\u2013679, URL https:\/\/www.usenix.org\/conference\/osdi23\/presentation\/li-zhouhan","key":"6605_CR22"},{"unstructured":"Libutti L, Igual FD, Pi\u00f1uel L, et\u00a0al (2020) Benchmarking Performance and Power of USB Accelerators for Inference with MLPerf. In: Workshop Accelerated Mach. Learn, p 1-15","key":"6605_CR23"},{"unstructured":"Ma X, Fang G, Wang X (2023) Llm-pruner: On the structural pruning of large language models. In: Oh A, Naumann T, Globerson A, et\u00a0al (eds) Advances in Neural Information Processing Systems, vol\u00a036. Curran Associates, Inc., pp 21702\u201321720, URL https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2023\/file\/44956951349095f74492a5471128a7e0-Paper-Conference.pdf","key":"6605_CR24"},{"key":"6605_CR25","doi-asserted-by":"publisher","DOI":"10.1109\/JIOT.2022.3176400","author":"P McEnroe","year":"2022","unstructured":"McEnroe P, Wang S, Liyanage M (2022) A survey on the convergence of edge computing and AI for UAVs: opportunities and challenges. IEEE Internet of Things J. https:\/\/doi.org\/10.1109\/JIOT.2022.3176400","journal-title":"IEEE Internet of Things J"},{"doi-asserted-by":"publisher","unstructured":"Mohammed T, Joe-Wong C, Babbar R, et\u00a0al (2020) Distributed inference acceleration with adaptive dnn partitioning and offloading. In: IEEE INFOCOM 2020-IEEE Conference on Computer Communications, IEEE, pp 854\u2013863, https:\/\/doi.org\/10.1109\/INFOCOM41043.2020.9155237","key":"6605_CR26","DOI":"10.1109\/INFOCOM41043.2020.9155237"},{"key":"6605_CR27","doi-asserted-by":"publisher","DOI":"10.1145\/3469029","author":"MGS Murshed","year":"2021","unstructured":"Murshed MGS, Murphy C, Hou D et al (2021) Machine learning at the network edge: a survey. ACM Comput Surv. https:\/\/doi.org\/10.1145\/3469029","journal-title":"ACM Comput Surv"},{"doi-asserted-by":"publisher","unstructured":"Narayanan D, Harlap A, Phanishayee A, et\u00a0al (2019) Pipedream: generalized pipeline parallelism for dnn training. In: Proceedings of the 27th ACM symposium on operating systems principles, pp 1\u201315, https:\/\/doi.org\/10.1145\/3341301.3359646","key":"6605_CR28","DOI":"10.1145\/3341301.3359646"},{"doi-asserted-by":"publisher","unstructured":"Nikoli\u0107 GS, Dimitrijevi\u0107 BR, Nikoli\u0107 TR, et\u00a0al (2022) A Survey of Three Types of Processing Units: CPU, GPU and TPU. In: Int. Scientific Conf. on Information, Communication and Energy Systems and Technologies (ICEST), pp 1\u20136, https:\/\/doi.org\/10.1109\/ICEST55168.2022.9828625","key":"6605_CR29","DOI":"10.1109\/ICEST55168.2022.9828625"},{"doi-asserted-by":"publisher","unstructured":"Parashar A, Abraham A, Chaudhary D, et\u00a0al (2020) Processor pipelining method for efficient deep neural network inference on embedded devices. In: 2020 IEEE 27th International Conference on High Performance Computing, Data, and Analytics (HiPC), IEEE, pp 82\u201390, https:\/\/doi.org\/10.1109\/HiPC50609.2020.00022","key":"6605_CR30","DOI":"10.1109\/HiPC50609.2020.00022"},{"key":"6605_CR31","first-page":"31","volume":"5","author":"P Raj","year":"2020","unstructured":"Raj P, Sekhar C (2020) Comparative Study on CPU, GPU and TPU. Int J Comput Sci and Inform Technol Edu 5:31\u201338","journal-title":"Int J Comput Sci and Inform Technol Edu"},{"doi-asserted-by":"publisher","unstructured":"Ren W, Qu Y, Dong C, et\u00a0al (2022) A Survey on Collaborative DNN Inference for Edge Intelligence. https:\/\/doi.org\/10.48550\/ARXIV.2207.07812, arXiv:2207.07812","key":"6605_CR32","DOI":"10.48550\/ARXIV.2207.07812"},{"unstructured":"Renda A, Frankle J, Carbin M (2020) Comparing rewinding and fine-tuning in neural network pruning. In: International Conference on Learning Representations","key":"6605_CR33"},{"unstructured":"Sedgewick R, Wayne KD (2011) Algorithms, 4th edn., Addison-Wesley Professional, pp 661\u2013666","key":"6605_CR34"},{"doi-asserted-by":"publisher","unstructured":"Seshadri K, Akin B, Laudon J, et\u00a0al (2021) An Evaluation of Edge TPU Accelerators for Convolutional Neural Networks. https:\/\/doi.org\/10.48550\/ARXIV.2102.10423, arXiv:2102.10423","key":"6605_CR35","DOI":"10.48550\/ARXIV.2102.10423"},{"doi-asserted-by":"publisher","unstructured":"Sun Y, Kist AM (2021) Deep Learning on Edge TPUs. https:\/\/doi.org\/10.48550\/ARXIV.2108.13732, arXiv:2108.13732","key":"6605_CR36","DOI":"10.48550\/ARXIV.2108.13732"},{"doi-asserted-by":"publisher","unstructured":"Thalluri LN, Venkat SN, Prasad CVVD, et\u00a0al (2021) Artificial Intelligence Enabled Smart City IoT System using Edge Computing. In: Int. Conf. on Smart Electronics and Communication (ICOSEC), pp 12\u201320, https:\/\/doi.org\/10.1109\/ICOSEC51865.2021.9591732","key":"6605_CR37","DOI":"10.1109\/ICOSEC51865.2021.9591732"},{"key":"6605_CR38","doi-asserted-by":"publisher","DOI":"10.1145\/3444692","author":"B Varghese","year":"2021","unstructured":"Varghese B, Wang N, Bermbach D et al (2021) A survey on edge performance benchmarking. ACM Comput Surv. https:\/\/doi.org\/10.1145\/3444692","journal-title":"ACM Comput Surv"},{"doi-asserted-by":"publisher","unstructured":"Villarrubia J, Costero L, Igual FD, et\u00a0al (2023) Improving inference time in multi-TPU systems with profiled model segmentation. In: 2023 31th Euromicro International Conference on Parallel, Distributed and Network-Based Processing (PDP), pp 84\u201391, https:\/\/doi.org\/10.1109\/PDP59025.2023.00020","key":"6605_CR39","DOI":"10.1109\/PDP59025.2023.00020"},{"key":"6605_CR40","doi-asserted-by":"publisher","DOI":"10.1109\/JIOT.2023.3323520","author":"L Wu","year":"2023","unstructured":"Wu L, Gao G, Yu J et al (2023) Pdd: partitioning dag-topology dnns for streaming tasks. IEEE Internet of Things J. https:\/\/doi.org\/10.1109\/JIOT.2023.3323520","journal-title":"IEEE Internet of Things J"},{"doi-asserted-by":"publisher","unstructured":"Xiang Y, Kim H (2019) Pipelined data-parallel cpu\/gpu scheduling for multi-dnn real-time inference. 2019 IEEE Real-Time Systems Symposium (RTSS) pp 392\u2013405. https:\/\/doi.org\/10.1109\/RTSS46320.2019.00042","key":"6605_CR41","DOI":"10.1109\/RTSS46320.2019.00042"},{"issue":"2","key":"6605_CR42","doi-asserted-by":"publisher","first-page":"595","DOI":"10.1109\/TNET.2020.3042320","volume":"29","author":"L Zeng","year":"2021","unstructured":"Zeng L, Chen X, Zhou Z et al (2021) Coedge: cooperative dnn inference with adaptive workload partitioning over heterogeneous edge devices. IEEE\/ACM Trans Network 29(2):595\u2013608. https:\/\/doi.org\/10.1109\/TNET.2020.3042320","journal-title":"IEEE\/ACM Trans Network"},{"issue":"2","key":"6605_CR43","doi-asserted-by":"publisher","first-page":"475","DOI":"10.1109\/TPDS.2022.3222509","volume":"34","author":"H Zhou","year":"2023","unstructured":"Zhou H, Li M, Wang N et al (2023) Accelerating deep learning inference via model parallelism and partial computation offloading. IEEE Trans Paral and Distribut Syst 34(2):475\u2013488. https:\/\/doi.org\/10.1109\/TPDS.2022.3222509","journal-title":"IEEE Trans Paral and Distribut Syst"},{"issue":"3","key":"6605_CR44","doi-asserted-by":"publisher","first-page":"825","DOI":"10.1109\/LWC.2019.2894703","volume":"8","author":"J Zhou","year":"2019","unstructured":"Zhou J, Wang Y, Ota K et al (2019) Aaiot: accelerating artificial intelligence in iot systems. IEEE Wireless Commun Lett 8(3):825\u2013828. https:\/\/doi.org\/10.1109\/LWC.2019.2894703","journal-title":"IEEE Wireless Commun Lett"},{"key":"6605_CR45","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11623","author":"Y Zhou","year":"2018","unstructured":"Zhou Y, Moosavi-Dezfooli SM, Cheung NM et al (2018) Adaptive quantization for deep neural network. Proceed AAAI Conference on Art Intell. https:\/\/doi.org\/10.1609\/aaai.v32i1.11623","journal-title":"Proceed AAAI Conference on Art Intell"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-024-06605-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11227-024-06605-9\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-024-06605-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,22]],"date-time":"2024-10-22T18:04:24Z","timestamp":1729620264000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11227-024-06605-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,22]]},"references-count":45,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2025,1]]}},"alternative-id":["6605"],"URL":"https:\/\/doi.org\/10.1007\/s11227-024-06605-9","relation":{},"ISSN":["0920-8542","1573-0484"],"issn-type":[{"type":"print","value":"0920-8542"},{"type":"electronic","value":"1573-0484"}],"subject":[],"published":{"date-parts":[[2024,10,22]]},"assertion":[{"value":"8 October 2024","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 October 2024","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"There are no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical approval"}}],"article-number":"60"}}