{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T23:46:30Z","timestamp":1783035990034,"version":"3.54.6"},"publisher-location":"New York, NY, USA","reference-count":54,"publisher":"ACM","funder":[{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["CNS-2149533"],"award-info":[{"award-number":["CNS-2149533"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["CNS-2421244"],"award-info":[{"award-number":["CNS-2421244"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,12,3]]},"DOI":"10.1145\/3769102.3772715","type":"proceedings-article","created":{"date-parts":[[2025,12,3]],"date-time":"2025-12-03T16:00:41Z","timestamp":1764777641000},"page":"1-16","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["SEEB-GPU: Early-Exit Aware Scheduling and Batching for Edge GPU Inference"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5848-5667","authenticated-orcid":false,"given":"Srinivasan","family":"Subramaniyan","sequence":"first","affiliation":[{"name":"Department of Electrical and Computer Engineering, The Ohio State University, Columbus, Ohio, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-0134-1082","authenticated-orcid":false,"given":"Rudra","family":"Joshi","sequence":"additional","affiliation":[{"name":"Department of Electrical and Computer Engineering, The Ohio State University, Columbus, Ohio, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9633-1418","authenticated-orcid":false,"given":"Xiaorui","family":"Wang","sequence":"additional","affiliation":[{"name":"Department of Electrical and Computer Engineering, The Ohio State University, Columbus, Ohio, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3603-1402","authenticated-orcid":false,"given":"Marco","family":"Brocanelli","sequence":"additional","affiliation":[{"name":"Department of Electrical and Computer Engineering, The Ohio State University, Columbus, Ohio, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,12,3]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.comcom.2021.01.021"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC41405.2020.00073"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"crossref","unstructured":"MR Ashuthosh Santosh Krishna Vishvas Sudarshan Srinivasan Sub-ramaniyan and Madhura Purnaprajna. 2022. MAPPARAT: A Resource Constrained FPGA-Based Accelerator for Sparse-Dense Matrix Multiplication. In 2022 35th International Conference on VLSI Design and 2022 21st International Conference on Embedded Systems (VLSID). IEEE Virtual Conference 102\u2013107.","DOI":"10.1109\/VLSID2022.2022.00031"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/RTAS58335.2023.00012"},{"key":"e_1_3_2_1_5_1","volume-title":"Latency-Guaranteed Co-Location of Inference and Training for Reducing Data Center Expenses. In 2024 IEEE 44th International Conference on Distributed Computing Systems (ICDCS). IEEE","author":"Chen Guoyu","year":"2024","unstructured":"Guoyu Chen, Srinivasan Subramaniyan, and Xiaorui Wang. 2024. Latency-Guaranteed Co-Location of Inference and Training for Reducing Data Center Expenses. In 2024 IEEE 44th International Conference on Distributed Computing Systems (ICDCS). IEEE, Jersey City, New Jersey, USA, 473\u2013484."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDCS54860.2022.00039"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW.2019.00363"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA56546.2023.10071121"},{"key":"e_1_3_2_1_9_1","volume-title":"Accessed","author":"Corrado Alessio","year":"2019","unstructured":"Alessio Corrado. 2019. Animals-10 dataset. https:\/\/www.kaggle.com\/datasets\/alessiocorrado99\/animals10. Contact: alessiocor-rado99@gmail.com. Accessed June 19, 2025."},{"key":"e_1_3_2_1_10_1","volume-title":"14th USENIX Symposium on Networked Systems Design and Implementation (NSDI 17)","author":"Crankshaw Daniel","year":"2017","unstructured":"Daniel Crankshaw, Xin Wang, Guilio Zhou, Michael J Franklin, Joseph E Gonzalez, and Ion Stoica. 2017. Clipper: A {Low-Latency} online prediction serving system. In 14th USENIX Symposium on Networked Systems Design and Implementation (NSDI 17). USENIX Association, Boston, MA, 613\u2013627."},{"key":"e_1_3_2_1_11_1","volume-title":"Exploiting Linear Structure Within Convolutional Networks for Efficient Evaluation. Advances in neural information processing systems 27","author":"Denton Emily","year":"2014","unstructured":"Emily Denton, Wojciech Zaremba, Joan Bruna, Yann LeCun, and Rob Fergus. 2014. Exploiting Linear Structure Within Convolutional Networks for Efficient Evaluation. Advances in neural information processing systems 27 (2014)."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3419111.3421284"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2015.2477042"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3517206.3526270"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIFS.2014.2359646"},{"key":"e_1_3_2_1_16_1","volume-title":"Deep learning-based image recognition for autonomous driving. IATSS research 43, 4","author":"Fujiyoshi Hironobu","year":"2019","unstructured":"Hironobu Fujiyoshi, Tsubasa Hirakawa, and Takayoshi Yamashita. 2019. Deep learning-based image recognition for autonomous driving. IATSS research 43, 4 (2019), 244\u2013252."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3529113.3529124"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.169"},{"key":"e_1_3_2_1_19_1","volume-title":"ICONIP 2013, daegu, korea, november 3\u20137, 2013. Proceedings, Part III 20","author":"Goodfellow Ian J","year":"2013","unstructured":"Ian J Goodfellow, Dumitru Erhan, Pierre Luc Carrier, Aaron Courville, Mehdi Mirza, Ben Hamner, Will Cukierski, Yichuan Tang, David Thaler, Dong-Hyun Lee, et al. 2013. Challenges in representation learning: A report on three machine learning contests. In Neural information processing: 20th international conference, ICONIP 2013, daegu, korea, november 3\u20137, 2013. Proceedings, Part III 20. Springer, Springer, Berlin, Heidelberg, 117\u2013124."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1002\/rob.21918"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3135974.3135993"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3643832.3661878"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3546192"},{"key":"e_1_3_2_1_24_1","volume-title":"The 2013 international joint conference on neural networks (IJCNN)","author":"Houben Sebastian","unstructured":"Sebastian Houben, Johannes Stallkamp, Jan Salmen, Marc Schlipsing, and Christian Igel. 2013. Detection of traffic signs in real-world images: The German Traffic Sign Detection Benchmark. In The 2013 international joint conference on neural networks (IJCNN). IEEE, Dallas, TX, USA, 1\u20138."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/3508391"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2020.3018269"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10586-017-1117-8"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3576842.3582375"},{"key":"e_1_3_2_1_29_1","unstructured":"Zhuang Liu Mingjie Sun Tinghui Zhou Gao Huang and Trevor Darrell. 2019. Rethinking the Value of Network Pruning. arXiv:1810.05270 [cs.LG] https:\/\/arxiv.org\/abs\/1810.05270"},{"key":"e_1_3_2_1_30_1","volume-title":"Power Capping of GPU Servers for Machine Learning Inference Optimization","author":"Ma Yuan","unstructured":"Yuan Ma, Srinivasan Subramaniyan, and Xiaorui Wang. 2025. Power Capping of GPU Servers for Machine Learning Inference Optimization. In ICPP. Association for Computing Machinery, San Dieogo, CA, USA."},{"key":"e_1_3_2_1_31_1","volume-title":"Hari Prabhat Gupta, and Tanima Dutta","author":"Mishra Rahul","year":"2020","unstructured":"Rahul Mishra, Hari Prabhat Gupta, and Tanima Dutta. 2020. A Survey on Deep Neural Network Compression: Challenges, Overview, and Solutions. arXiv:2010.03954 [cs.LG] https:\/\/arxiv.org\/abs\/2010.03954"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2022.3144614"},{"key":"e_1_3_2_1_33_1","unstructured":"NVIDIA. 2024. Triton Inference Server. https:\/\/github.com\/triton-inference-server\/server"},{"key":"e_1_3_2_1_34_1","volume-title":"Accessed","author":"NVIDIA.","year":"2025","unstructured":"NVIDIA. 2025. NVIDIA Jetson AGX Orin Technical Brief. https:\/\/www.nvidia.com\/content\/dam\/en-zz\/Solutions\/gtcf21\/jetson-orin\/nvidia-jetson-agx-orin-technical-brief.pdf. Accessed: June 20, 2025."},{"key":"e_1_3_2_1_35_1","unstructured":"NVIDIA Corporation. 2020. NVIDIA A100 Tensor Core GPU Datasheet. https:\/\/www.nvidia.com\/content\/dam\/en-zz\/Solutions\/Data-Center\/a100\/pdf\/nvidia-a100-datasheet-us-nvidia-1758950-r4-web.pdf. Accessed: 2025-06-20."},{"key":"e_1_3_2_1_36_1","unstructured":"Christopher Olston Noah Fiedel Kiril Gorovoy Jeremiah Harmsen Li Lao Fangwei Li Vinu Rajashekhar Sukriti Ramesh and Jordan Soyke. 2017. TensorFlow-Serving: Flexible High-Performance ML Serving. arXiv:1712.06139 [cs.DC] https:\/\/arxiv.org\/abs\/1712.06139"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/3453417.3453432"},{"key":"e_1_3_2_1_38_1","unstructured":"Antonio Polino Razvan Pascanu and Dan Alistarh. 2018. Model compression via distillation and quantization. arXiv:1802.05668 [cs.NE] https:\/\/arxiv.org\/abs\/1802.05668"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/3698767"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/VLSID51830.2021.00046"},{"key":"e_1_3_2_1_41_1","unstructured":"NVIDIA Triton Inference Server. 2021. Triton inference server."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPSW59300.2023.00071"},{"key":"e_1_3_2_1_43_1","volume-title":"Proceedings of the IEEE International Performance, Computing, and Communications Conference (IPCCC). IEEE","author":"Subramaniyan Srinivasan","year":"2025","unstructured":"Srinivasan Subramaniyan and Xiaorui Wang. 2025. Exploiting ML Task Correlation in the Minimization of Capital Expense for GPU Data Centers. In Proceedings of the IEEE International Performance, Computing, and Communications Conference (IPCCC). IEEE, Austin Tx, USA."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/3761812"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICPR.2016.7900006"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICACCS48705.2020.9074444"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/RTSS55097.2022.00042"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/RTSS52674.2021.00021"},{"key":"e_1_3_2_1_49_1","volume-title":"19th USENIX Symposium on Networked Systems Design and Implementation (NSDI 22)","author":"Weng Qizhen","year":"2022","unstructured":"Qizhen Weng, Wencong Xiao, Yinghao Yu, Wei Wang, Cheng Wang, Jian He, Yong Li, Liping Zhang, Wei Lin, and Yu Ding. 2022. {MLaaS} in the wild: Workload analysis and scheduling in {Large-Scale} heterogeneous {GPU} clusters. In 19th USENIX Symposium on Networked Systems Design and Implementation (NSDI 22). USENIX Association, Renton, WA, USA, 945\u2013960."},{"key":"e_1_3_2_1_50_1","volume-title":"2021 IEEE\/ACM Symposium on Edge Computing (SEC). IEEE","author":"Yang Zhe","year":"2021","unstructured":"Zhe Yang, Klara Nahrstedt, Hongpeng Guo, and Qian Zhou. 2021. Deeprt: A soft real time scheduler for computer vision applications on the edge. In 2021 IEEE\/ACM Symposium on Edge Computing (SEC). IEEE, San Jose, CA, USA, 271\u2013284."},{"key":"e_1_3_2_1_51_1","volume-title":"Model-Switching: Dealing with Fluctuating Workloads in Machine-Learning-as-a-Service Systems. In 12th USENIX Workshop on Hot Topics in Cloud Computing (HotCloud 20)","author":"Zhang Jeff","year":"2020","unstructured":"Jeff Zhang, Sameh Elnikety, Shuayb Zarar, Atul Gupta, and Siddharth Garg. 2020. Model-Switching: Dealing with Fluctuating Workloads in Machine-Learning-as-a-Service Systems. In 12th USENIX Workshop on Hot Topics in Cloud Computing (HotCloud 20). USENIX Association, virtual event."},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/WCNC57260.2024.10571127"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/TNSM.2024.3409701"},{"key":"e_1_3_2_1_54_1","unstructured":"Michael Zhu and Suyog Gupta. 2017. To prune or not to prune: exploring the efficacy of pruning for model compression. arXiv:1710.01878 [stat.ML] https:\/\/arxiv.org\/abs\/1710.01878"}],"event":{"name":"SEC '25: Tenth ACM\/IEEE Symposium on Edge Computing","location":"the Hilton Arlington National Landing Arlington VA USA","acronym":"SEC '25","sponsor":["SIGMOBILE ACM Special Interest Group on Mobility of Systems, Users, Data and Computing","IEEE Computer Society"]},"container-title":["Proceedings of the Tenth ACM\/IEEE Symposium on Edge Computing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3769102.3772715","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,3]],"date-time":"2025-12-03T16:01:57Z","timestamp":1764777717000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3769102.3772715"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,3]]},"references-count":54,"alternative-id":["10.1145\/3769102.3772715","10.1145\/3769102"],"URL":"https:\/\/doi.org\/10.1145\/3769102.3772715","relation":{},"subject":[],"published":{"date-parts":[[2025,12,3]]},"assertion":[{"value":"2025-12-03","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}