{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,5]],"date-time":"2026-03-05T15:46:04Z","timestamp":1772725564973,"version":"3.50.1"},"reference-count":47,"publisher":"Springer Science and Business Media LLC","issue":"13","license":[{"start":{"date-parts":[[2023,4,4]],"date-time":"2023-04-04T00:00:00Z","timestamp":1680566400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,4,4]],"date-time":"2023-04-04T00:00:00Z","timestamp":1680566400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"published-print":{"date-parts":[[2023,9]]},"DOI":"10.1007\/s11227-023-05183-6","type":"journal-article","created":{"date-parts":[[2023,4,4]],"date-time":"2023-04-04T09:13:20Z","timestamp":1680599600000},"page":"14172-14199","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["CoFB: latency-constrained co-scheduling of flows and batches for deep learning inference service on the CPU\u2013GPU system"],"prefix":"10.1007","volume":"79","author":[{"given":"Qi","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yi","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tao","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Depei","family":"Qian","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,4,4]]},"reference":[{"key":"5183_CR1","unstructured":"Belay A, Prekas G, Klimovic A, et\u00a0al (2014) IX: a protected dataplane operating system for high throughput and low latency. In: 11th USENIX Symposium on Operating Systems Design and Implementation (OSDI 14), pp 49\u201365"},{"key":"5183_CR2","unstructured":"Bellis R (2017) A tool for measuring NIC utilisation. https:\/\/urldefense.com\/v3\/__https:\/\/github.com\/isc-projects\/ethq__;!!NLFGqXoFfo8MMQ!rvxSn_ME1CVNIBsOZ_OWUWuC_bWeRcwwS5qNRm0hmoQZLvJxrQSGicDt07eyidl87lOT9k0Dw8SgkC7bvw$. Accessed 2022"},{"key":"5183_CR3","unstructured":"Burns E (2017) How facebook uses deep learning models to engage users. https:\/\/www.techtarget.com\/searchenterpriseai\/news\/450420189\/How-Facebook-uses-deep-learning-models-to-engage-users. Accessed 2022"},{"key":"5183_CR4","unstructured":"Chen T, Moreau T, Jiang Z, et\u00a0al (2018) TVM: an automated end-to-end optimizing compiler for deep learning. In: 13th USENIX Symposium on Operating Systems Design and Implementation (OSDI 18), pp 578\u2013594"},{"key":"5183_CR5","doi-asserted-by":"crossref","unstructured":"Choi Y, Rhu M (2020) Prema: a predictive multi-task scheduling algorithm for preemptible neural processing units. In: 2020 IEEE International Symposium on High Performance Computer Architecture (HPCA). IEEE, pp 220\u2013233","DOI":"10.1109\/HPCA47549.2020.00027"},{"key":"5183_CR6","doi-asserted-by":"crossref","unstructured":"Choi Y, Kim Y, Rhu M (2021) LazyBatching: an SLA-aware batching system for cloud machine learning inference. In: 2021 IEEE International Symposium on High-Performance Computer Architecture (HPCA). IEEE, pp 493\u2013506","DOI":"10.1109\/HPCA51647.2021.00049"},{"key":"5183_CR7","unstructured":"Corportion N (2019) Nvidia tensor cores-unprecedented acceleration for HPC and AI. https:\/\/www.nvidia.com\/en-us\/data-center\/tensor-cores\/. Accessed 2022"},{"key":"5183_CR8","unstructured":"Crankshaw D, Wang X, Zhou G et\u00a0al (2017) Clipper: a low-latency online prediction serving system. In: 14th USENIX Symposium on Networked Systems Design and Implementation (NSDI 17), pp 613\u2013627"},{"key":"5183_CR9","doi-asserted-by":"crossref","unstructured":"Cui W, Wei M, Chen Q et\u00a0al (2019) Ebird: elastic batch for improving responsiveness and throughput of deep learning services. In: 2019 IEEE 37th International Conference on Computer Design (ICCD). IEEE, pp 497\u2013505","DOI":"10.1109\/ICCD46524.2019.00075"},{"issue":"6","key":"5183_CR10","doi-asserted-by":"publisher","first-page":"1307","DOI":"10.1109\/TPDS.2020.3047638","volume":"32","author":"W Cui","year":"2020","unstructured":"Cui W, Chen Q, Zhao H et al (2020) E2 bird: enhanced elastic batch for improving responsiveness and throughput of deep learning services. IEEE Trans Parallel Distrib Syst 32(6):1307\u20131321","journal-title":"IEEE Trans Parallel Distrib Syst"},{"key":"5183_CR11","doi-asserted-by":"crossref","unstructured":"Cui W, Zhao H, Chen Q et\u00a0al (2021) Enable simultaneous DNN services based on deterministic operator overlap and precise latency prediction. In: Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis, pp 1\u201315","DOI":"10.1145\/3458817.3476143"},{"key":"5183_CR12","unstructured":"Cui W, Zhao H, Chen Q et\u00a0al (2022) DVABatch: diversity-aware multi-entrymulti-exit batching for efficient processing of DNN services on GPUs. In: 2022 USENIX Annual Technical Conference (USENIX ATC 22), pp 183\u2013198"},{"key":"5183_CR13","unstructured":"Devlin J, Chang MW, Lee K et\u00a0al (2018) Bert: pre-training of deep bidirectional transformers for language understanding. ArXiv:https:\/\/arxiv.org\/abs\/1810.04805"},{"key":"5183_CR14","doi-asserted-by":"crossref","unstructured":"Du J, Jiang J, You Y et\u00a0al (2022) Handling heavy-tailed input of transformer inference on GPUS. In: Proceedings of the 36th ACM International Conference on Supercomputing, pp 1\u201311","DOI":"10.1145\/3524059.3532372"},{"key":"5183_CR15","doi-asserted-by":"crossref","unstructured":"Fang J, Yu Y, Zhao C et\u00a0al (2021) Turbotransformers: an efficient GPU serving system for transformer models. In: Proceedings of the 26th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming, pp 389\u2013402","DOI":"10.1145\/3437801.3441578"},{"key":"5183_CR16","unstructured":"Fried J, Ruan Z, Ousterhout A et\u00a0al (2020) Caladan: mitigating interference at microsecond timescales. In: 14th USENIX Symposium on Operating Systems Design and Implementation (OSDI 20), pp 281\u2013297"},{"key":"5183_CR17","doi-asserted-by":"crossref","unstructured":"Gao P, Yu L, Wu Y et\u00a0al (2018) Low latency RNN inference with cellular batching. In: Proceedings of the Thirteenth EuroSys Conference, pp 1\u201315","DOI":"10.1145\/3190508.3190541"},{"issue":"5","key":"5183_CR18","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3495532","volume":"27","author":"Y Gong","year":"2022","unstructured":"Gong Y, Yuan G, Zhan Z et al (2022) Automatic mapping of the best-suited DNN pruning schemes for real-time mobile acceleration. ACM Trans Des Autom Electron Syst (TODAES) 27(5):1\u201326","journal-title":"ACM Trans Des Autom Electron Syst (TODAES)"},{"key":"5183_CR19","unstructured":"Gujarati A, Karimi R, Alzayat S et\u00a0al (2020) Serving DNNs like clockwork: performance predictability from the bottom up. In: 14th USENIX Symposium on Operating Systems Design and Implementation (OSDI 20), pp 443\u2013462"},{"key":"5183_CR20","doi-asserted-by":"crossref","unstructured":"Gupta U, Hsia S, Saraph V et\u00a0al (2020) Deeprecsys: a system for optimizing end-to-end at-scale neural recommendation inference. In: 2020 ACM\/IEEE 47th Annual International Symposium on Computer Architecture (ISCA). IEEE, pp 982\u2013995","DOI":"10.1109\/ISCA45697.2020.00084"},{"key":"5183_CR21","doi-asserted-by":"crossref","unstructured":"Hanson E, Li S, Li H et\u00a0al (2022) Cascading structured pruning: enabling high data reuse for sparse dnn accelerators. In: Proceedings of the 49th Annual International Symposium on Computer Architecture, pp 522\u2013535","DOI":"10.1145\/3470496.3527419"},{"key":"5183_CR22","doi-asserted-by":"crossref","unstructured":"Hazelwood K, Bird S, Brooks D et\u00a0al (2018) Applied machine learning at facebook: a datacenter infrastructure perspective. In: 2018 IEEE International Symposium on High Performance Computer Architecture (HPCA). IEEE, pp 620\u2013629","DOI":"10.1109\/HPCA.2018.00059"},{"key":"5183_CR23","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S et\u00a0al (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"5183_CR24","unstructured":"Jain H, Verma T, Tabrizian I et\u00a0al (2019) Nvidia triton inference server. https:\/\/github.com\/triton-inference-server\/server. Accessed 2022"},{"key":"5183_CR25","unstructured":"Jeong E, Wood S, Jamshed M et\u00a0al (2014) mTCP: a highly scalable user-level TCP stack for multicore systems. In: 11th USENIX Symposium on Networked Systems Design and Implementation (NSDI 14), pp 489\u2013502"},{"key":"5183_CR26","unstructured":"Kaffes K, Chong T, Humphries JT et\u00a0al (2019) Shinjuku: preemptive scheduling for microsecond-scale tail latency. In: 16th USENIX Symposium on Networked Systems Design and Implementation (NSDI 19), pp 345\u2013360"},{"key":"5183_CR27","unstructured":"Kiril\u00a0Gorovoy eaChristopher\u00a0Olston (2016) Tensorflow serving. https:\/\/github.com\/tensorflow\/serving. Accessed 2022"},{"key":"5183_CR28","doi-asserted-by":"crossref","unstructured":"Kriman S, Beliaev S, Ginsburg B et al (2020) Quartznet: deep automatic speech recognition with 1d time-channel separable convolutions. In: ICASSP 2020\u20132020 IEEE International Conference on Acoustics. Speech and Signal Processing (ICASSP). IEEE, pp 6124\u20136128","DOI":"10.1109\/ICASSP40776.2020.9053889"},{"key":"5183_CR29","unstructured":"Lee Y, Scolari A, Chun BG et\u00a0al (2018) PRETZEL: opening the black box of machine learning prediction serving systems. In: 13th USENIX Symposium on Operating Systems Design and Implementation (OSDI 18), pp 611\u2013626"},{"key":"5183_CR30","doi-asserted-by":"crossref","unstructured":"Li Y, Han Z, Zhang Q et\u00a0al (2020) Automating cloud deployment for deep learning inference of real-time online services. In: IEEE INFOCOM 2020-IEEE Conference on Computer Communications. IEEE, pp 1668\u20131677","DOI":"10.1109\/INFOCOM41043.2020.9155267"},{"key":"5183_CR31","unstructured":"Ma L, Xie Z, Yang Z, et\u00a0al (2020) Rammer: enabling holistic deep learning compiler optimizations with rTasks. In: 14th USENIX Symposium on Operating Systems Design and Implementation (OSDI 20), pp 881\u2013897"},{"key":"5183_CR32","unstructured":"Nik K, GSDavid\u00a0Beer (2019) Nvidia data center GPU manager (DCGM). https:\/\/github.com\/NVIDIA\/DCGM. Accessed 2022"},{"key":"5183_CR33","unstructured":"Ousterhout A, Fried J, Behrens J et\u00a0al (2019) Shenango: achieving high CPU efficiency for latency-sensitive datacenter workloads. In: 16th USENIX Symposium on Networked Systems Design and Implementation (NSDI 19), pp 361\u2013378"},{"key":"5183_CR34","doi-asserted-by":"crossref","unstructured":"Prekas G, Kogias M, Bugnion E (2017) Zygos: achieving low tail latency for microsecond-scale networked tasks. In: Proceedings of the 26th Symposium on Operating Systems Principles, pp 325\u2013341","DOI":"10.1145\/3132747.3132780"},{"key":"5183_CR35","unstructured":"Qin H, Li Q, Speiser J et\u00a0al (2018) Arachne: core-aware thread management. In: 13th USENIX Symposium on Operating Systems Design and Implementation (OSDI 18), pp 145\u2013160"},{"key":"5183_CR36","doi-asserted-by":"crossref","unstructured":"Qin Z, Wang H, Li X (2020) Ultra fast structure-aware deep lane detection. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XXIV 16. Springer, pp 276\u2013291","DOI":"10.1007\/978-3-030-58586-0_17"},{"key":"5183_CR37","first-page":"1","volume":"10","author":"N Rathi","year":"2021","unstructured":"Rathi N, Roy K (2021) DIET-SNN: a low-latency spiking neural network with direct input encoding and leakage and threshold optimization. IEEE Trans Neural Netw Learn Syst 10:1\u20139","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"key":"5183_CR38","doi-asserted-by":"crossref","unstructured":"Shen H, Chen L, Jin Y et\u00a0al (2019) Nexus: a GPU cluster engine for accelerating DNN-based video analysis. In: Proceedings of the 27th ACM Symposium on Operating Systems Principles, pp 322\u2013337","DOI":"10.1145\/3341301.3359658"},{"key":"5183_CR39","unstructured":"Shen H, Roesch J, Chen Z et al (2021) Nimble: efficiently compiling dynamic neural networks for model inference. In: Proceedings of Machine Learning and Systems, vol 3, pp 208\u2013222"},{"key":"5183_CR40","unstructured":"Tanmay\u00a0V, eaHemant\u00a0J (2022) triton-inference-server-client. https:\/\/github.com\/triton-inference-server\/client. Accessed 2022"},{"key":"5183_CR41","doi-asserted-by":"crossref","unstructured":"Wang Z, Wei Y, Lee M et\u00a0al (2022) Merlin HugeCTR: GPU-accelerated recommender system training and inference. In: Proceedings of the 16th ACM Conference on Recommender Systems, pp 534\u2013537","DOI":"10.1145\/3523227.3547405"},{"key":"5183_CR42","unstructured":"Weng O (2021) Neural network quantization for efficient inference: A survey. arXiv preprint arXiv:2112.06126"},{"key":"5183_CR43","doi-asserted-by":"crossref","unstructured":"Yang Y, Zhao L, Li Y et\u00a0al (2022) INFless: a native serverless system for low-latency, high-throughput inference. In: Proceedings of the 27th ACM International Conference on Architectural Support for Programming Languages and Operating Systems, pp 768\u2013781","DOI":"10.1145\/3503222.3507709"},{"key":"5183_CR44","doi-asserted-by":"crossref","unstructured":"Yeh TT, Sinclair MD, Beckmann BM et\u00a0al (2021) Deadline-aware offloading for high-throughput accelerators. In: 2021 IEEE International Symposium on High-Performance Computer Architecture (HPCA). IEEE, pp 479\u2013492","DOI":"10.1109\/HPCA51647.2021.00048"},{"key":"5183_CR45","unstructured":"Yu GI, Jeong JS, Kim GW et\u00a0al (2022) Orca: a distributed serving system for Transformer-Based generative models. In: 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22), pp 521\u2013538"},{"key":"5183_CR46","first-page":"1","volume":"01","author":"Q Zhang","year":"2022","unstructured":"Zhang Q, Cheng X, Chen Y et al (2022) Quantifying the knowledge in a DNN to explain knowledge distillation for classification. IEEE Trans Pattern Anal Mach Intell 01:1\u201317","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"5183_CR47","unstructured":"Zhu H, Wu R, Diao Y et\u00a0al (2022) ROLLER: fast and efficient tensor compilation for deep learning. In: 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22), pp 233\u2013248"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-023-05183-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11227-023-05183-6","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-023-05183-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,8]],"date-time":"2026-01-08T07:08:09Z","timestamp":1767856089000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11227-023-05183-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,4,4]]},"references-count":47,"journal-issue":{"issue":"13","published-print":{"date-parts":[[2023,9]]}},"alternative-id":["5183"],"URL":"https:\/\/doi.org\/10.1007\/s11227-023-05183-6","relation":{},"ISSN":["0920-8542","1573-0484"],"issn-type":[{"value":"0920-8542","type":"print"},{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,4,4]]},"assertion":[{"value":"7 March 2023","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 April 2023","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}