{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,19]],"date-time":"2026-01-19T08:29:44Z","timestamp":1768811384446,"version":"3.49.0"},"publisher-location":"New York, NY, USA","reference-count":38,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,8,7]],"date-time":"2023-08-07T00:00:00Z","timestamp":1691366400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"China University Industry Research Innovation Foundation","award":["No. 2021FNA04005"],"award-info":[{"award-number":["No. 2021FNA04005"]}]},{"name":"National Science Foundation of China","award":["No. 61832005"],"award-info":[{"award-number":["No. 61832005"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,8,7]]},"DOI":"10.1145\/3605573.3605615","type":"proceedings-article","created":{"date-parts":[[2023,9,13]],"date-time":"2023-09-13T16:21:16Z","timestamp":1694622076000},"page":"72-81","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":7,"title":["BIRP: Batch-aware Inference Workload Redistribution and Parallel Scheme for Edge Collaboration"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-5246-6018","authenticated-orcid":false,"given":"Hesheng","family":"Sun","sequence":"first","affiliation":[{"name":"Nanjing University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-8927-7115","authenticated-orcid":false,"given":"Xinyi","family":"Chen","sequence":"additional","affiliation":[{"name":"Nanjing University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1625-7575","authenticated-orcid":false,"given":"Zhuzhong","family":"Qian","sequence":"additional","affiliation":[{"name":"Nanjing University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-3895-8189","authenticated-orcid":false,"given":"Zengji","family":"Li","sequence":"additional","affiliation":[{"name":"Nanjing University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0722-1757","authenticated-orcid":false,"given":"Ning","family":"Chen","sequence":"additional","affiliation":[{"name":"Nanjing University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0846-0522","authenticated-orcid":false,"given":"Tuo","family":"Cao","sequence":"additional","affiliation":[{"name":"Nanjing University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-8856-2471","authenticated-orcid":false,"given":"Suwei","family":"Xu","sequence":"additional","affiliation":[{"name":"Nanjing University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-4202-7793","authenticated-orcid":false,"given":"Yitong","family":"Zhou","sequence":"additional","affiliation":[{"name":"Nanjing University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,9,13]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/JSAC.2019.2894306"},{"key":"e_1_3_2_1_2_1","unstructured":"Alexey Bochkovskiy Chien-Yao Wang and Hong-Yuan\u00a0Mark Liao. 2020. YOLOv4: Optimal Speed and Accuracy of Object Detection. arxiv:2004.10934\u00a0[cs eess]"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jpdc.2022.03.010"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.2019.2921977"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.comnet.2021.108186"},{"key":"e_1_3_2_1_6_1","volume-title":"Serving Heterogeneous Machine Learning Models on Multi-GPU Servers with Spatio-Temporal Sharing. In 2022 USENIX Annual Technical Conference (USENIX ATC 22)","author":"Choi Seungbeom","year":"2022","unstructured":"Seungbeom Choi, Sunho Lee, Yeonjae Kim, Jongse Park, Youngjin Kwon, and Jaehyuk Huh. 2022. Serving Heterogeneous Machine Learning Models on Multi-GPU Servers with Spatio-Temporal Sharing. In 2022 USENIX Annual Technical Conference (USENIX ATC 22). USENIX Association, Carlsbad, CA, 199\u2013216."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/SECON55815.2022.9918592"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/JIOT.2020.2984887"},{"key":"e_1_3_2_1_9_1","volume-title":"BERT: Pre-Training of Deep Bidirectional Transformers for Language Understanding. arxiv:1810.04805\u00a0[cs]","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. BERT: Pre-Training of Deep Bidirectional Transformers for Language Understanding. arxiv:1810.04805\u00a0[cs]"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2021.3112604"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/MIE.2020.3026837"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3097983.3098043"},{"key":"e_1_3_2_1_13_1","unstructured":"Google. 2019. Edge TPU. https:\/\/cloud.google.com\/edge-tpu"},{"key":"e_1_3_2_1_14_1","unstructured":"Gurobi. 2023. Gurobi Optimization. https:\/\/www.gurobi.com\/"},{"key":"e_1_3_2_1_15_1","volume-title":"Microsecond-scale Preemption for Concurrent GPU-accelerated DNN Inferences. In 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22)","author":"Han Mingcong","year":"2022","unstructured":"Mingcong Han, Hanze Zhang, Rong Chen, and Haibo Chen. 2022. Microsecond-scale Preemption for Concurrent GPU-accelerated DNN Inferences. In 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22). USENIX Association, Carlsbad, CA, 539\u2013558."},{"key":"e_1_3_2_1_16_1","unstructured":"Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2015. Deep Residual Learning for Image Recognition. arxiv:1512.03385\u00a0[cs]"},{"key":"e_1_3_2_1_17_1","unstructured":"Huawei. 2019. ATC Tool - Ascend Data Center Solution V100R020C30 Center Inference Solution Description 01 - Huawei. https:\/\/e.huawei.com\/hk\/material\/"},{"key":"e_1_3_2_1_18_1","unstructured":"Huawei. 2019. Huawei Ascend. https:\/\/www.hiascend.com\/"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/SECON48991.2020.9158425"},{"key":"e_1_3_2_1_20_1","volume-title":"Convolutional Networks for Images, Speech, and Time-Series. The handbook of brain theory and neural networks 3361, 10","author":"LeCun Yann","year":"1995","unstructured":"Yann LeCun, Yoshua Bengio, and T\u00a0Bell Laboratories. 1995. Convolutional Networks for Images, Speech, and Time-Series. The handbook of brain theory and neural networks 3361, 10 (1995), 255\u2013258."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/COMST.2021.3106401"},{"key":"e_1_3_2_1_22_1","unstructured":"Nivdia. 2019. TensorRT SDK | NVIDIA Developer. https:\/\/developer.nvidia.com\/tensorrt"},{"key":"e_1_3_2_1_23_1","unstructured":"Nvidia. 2019. Jetson Nano Developer Kit. https:\/\/developer.nvidia.com\/embedded\/jetson-nano-developer-kit"},{"key":"e_1_3_2_1_24_1","unstructured":"Nvidia. 2022. Design & Professional Visualization Solutions | NVIDIA. https:\/\/www.nvidia.com\/en-us\/design-visualization\/nvlink-bridges\/"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","unstructured":"George Plastiras Maria Terzi Christos Kyrkou and Theocharis Theocharidcs. 2018. Edge Intelligence: Challenges and Opportunities of Near-Sensor Machine Learning Applications. In 2018 IEEE 29th International Conference on Application-Specific Systems Architectures and Processors (ASAP). 1\u20137. https:\/\/doi.org\/10.1109\/ASAP.2018.8445118","DOI":"10.1109\/ASAP.2018.8445118"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPEC43674.2020.9286149"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPEC49654.2021.9622867"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPEC55821.2022.9926331"},{"key":"e_1_3_2_1_29_1","volume-title":"INFaaS: Automated Model-less Inference Serving. In 2021 USENIX Annual Technical Conference (USENIX ATC 21)","author":"Romero Francisco","year":"2021","unstructured":"Francisco Romero, Qian Li, Neeraja\u00a0J. Yadwadkar, and Christos Kozyrakis. 2021. INFaaS: Automated Model-less Inference Serving. In 2021 USENIX Annual Technical Conference (USENIX ATC 21). USENIX Association, 397\u2013411."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/TNSM.2019.2937342"},{"key":"e_1_3_2_1_31_1","volume-title":"Proceedings of the 13th Usenix Conference on Networked Systems Design and Implementation","author":"Venkataraman Shivaram","year":"2016","unstructured":"Shivaram Venkataraman, Zongheng Yang, Michael Franklin, Benjamin Recht, and Ion Stoica. 2016. Ernest: Efficient Performance Prediction for Large-Scale Advanced Analytics. In Proceedings of the 13th Usenix Conference on Networked Systems Design and Implementation (Santa Clara, CA) (NSDI\u201916). USENIX Association, USA, 363\u2013378."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/3472883.3486987"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM48880.2022.9796961"},{"key":"e_1_3_2_1_34_1","volume-title":"19th USENIX Symposium on Networked Systems Design and Implementation (NSDI 22)","author":"Weng Qizhen","year":"2022","unstructured":"Qizhen Weng, Wencong Xiao, Yinghao Yu, Wei Wang, Cheng Wang, Jian He, Yong Li, Liping Zhang, Wei Lin, and Yu Ding. 2022. MLaaS in the wild: Workload analysis and scheduling in Large-Scale heterogeneous GPU clusters. In 19th USENIX Symposium on Networked Systems Design and Implementation (NSDI 22). USENIX Association, 945\u2013960."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.2021.3119950"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3458864.3467882"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.sysarc.2022.102636"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.2019.2918951"}],"event":{"name":"ICPP 2023: 52nd International Conference on Parallel Processing","location":"Salt Lake City UT USA","acronym":"ICPP 2023"},"container-title":["Proceedings of the 52nd International Conference on Parallel Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3605573.3605615","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3605573.3605615","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T17:49:04Z","timestamp":1750182544000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3605573.3605615"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,8,7]]},"references-count":38,"alternative-id":["10.1145\/3605573.3605615","10.1145\/3605573"],"URL":"https:\/\/doi.org\/10.1145\/3605573.3605615","relation":{},"subject":[],"published":{"date-parts":[[2023,8,7]]},"assertion":[{"value":"2023-09-13","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}