{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,19]],"date-time":"2026-06-19T13:08:15Z","timestamp":1781874495509,"version":"3.54.5"},"reference-count":33,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"4","license":[{"start":{"date-parts":[[2025,7,1]],"date-time":"2025-07-01T00:00:00Z","timestamp":1751328000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,7,1]],"date-time":"2025-07-01T00:00:00Z","timestamp":1751328000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,7,1]],"date-time":"2025-07-01T00:00:00Z","timestamp":1751328000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"Grant-in-Aid for Japan Society for the Promotion of Science (JSPS) Fellows","award":["23KJ1786"],"award-info":[{"award-number":["23KJ1786"]}]},{"name":"Japan Science and Technology Agency (JST) PRESTO","award":["23828673"],"award-info":[{"award-number":["23828673"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62471383"],"award-info":[{"award-number":["62471383"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Sustain. Comput."],"published-print":{"date-parts":[[2025,7]]},"DOI":"10.1109\/tsusc.2025.3528105","type":"journal-article","created":{"date-parts":[[2025,1,10]],"date-time":"2025-01-10T15:56:36Z","timestamp":1736524596000},"page":"678-689","source":"Crossref","is-referenced-by-count":3,"title":["Serving Transformer Models via Joint Requst Scheduling and Batching in the Network Edge"],"prefix":"10.1109","volume":"10","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-2824-7927","authenticated-orcid":false,"given":"Boqian","family":"Fu","sequence":"first","affiliation":[{"name":"School of Computer Science and Engineering, University of Aizu, Aizuwakamatsu, Japan"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Fahao","family":"Chen","sequence":"additional","affiliation":[{"name":"School of Computer Science and Engineering, University of Aizu, Aizuwakamatsu, Japan"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5303-0700","authenticated-orcid":false,"given":"Peng","family":"Li","sequence":"additional","affiliation":[{"name":"School of Cyber Science and Engineering, Xi&#x2019;an Jiaotong University, Xi&#x2019;an, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3276-1202","authenticated-orcid":false,"given":"Deze","family":"Zeng","sequence":"additional","affiliation":[{"name":"School of Computer Science, China University of Geosciences, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref2","article-title":"BERT: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2018"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/W14-4012"},{"key":"ref5","first-page":"14011","article-title":"Accelerating training of transformer-based language models with progressive layer dropping","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Zhang"},{"key":"ref6","first-page":"1","article-title":"ET: Re-thinking self-attention for transformer models on GPUs","volume-title":"Proc. Int. Conf. High Perform. Comput. Netw. Storage Anal.","author":"Chen"},{"key":"ref7","first-page":"711","article-title":"Data movement is all you need: A case study on optimizing transformers","volume-title":"Proc. 4th Conf. Mach. Learn. Syst.","author":"Ivanov"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1145\/3437801.3441578"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.417"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W18-5446"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICCD46524.2019.00075"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1145\/3437984.3458837"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1145\/3503222.3507723"},{"key":"ref14","first-page":"613","article-title":"Clipper: A low-latency online prediction serving system","volume-title":"Proc. 14th USENIX Symp. Netw. Syst. Des. Implementation","author":"Crankshaw"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA51647.2021.00049"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1145\/3545008.3545052"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1145\/3447548.3467146"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1145\/3605573.3605598"},{"key":"ref19","first-page":"1877","article-title":"Language models are few-shot learners","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Brown"},{"key":"ref20","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","author":"Raffel","year":"2019"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.3115\/1119176.1119195"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D13-1170"},{"key":"ref23","first-page":"265","article-title":"TensorFlow: A system for large-scale machine learning","volume-title":"Proc. 12th USENIX Symp. Operating Syst. Des. Implementation","author":"Abadi"},{"key":"ref24","first-page":"8026","article-title":"PyTorch: An imperative style, high-performance deep learning library","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Paszke"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA52012.2021.00060"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/DAC18074.2021.9586134"},{"key":"ref27","first-page":"11648","article-title":"EL-attention: Memory efficient lossless attention for generation","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Yan"},{"key":"ref28","first-page":"6543","article-title":"TeraPipe: Token-level pipeline parallelism for training large-scale language models","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Li"},{"key":"ref29","first-page":"1","article-title":"Efficient large-scale language model training on GPU clusters using Megatron-LM","volume-title":"Proc. Int. Conf. High Perform. Comput. Netw. Storage Anal.","author":"Narayanan"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM41043.2020.9155267"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476168"},{"key":"ref32","first-page":"8343","article-title":"Nimble: Lightweight and parallel GPU task scheduling for deep learning","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Kwon"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA47549.2020.00027"}],"container-title":["IEEE Transactions on Sustainable Computing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/7274860\/11121336\/10836911.pdf?arnumber=10836911","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,9]],"date-time":"2025-08-09T04:45:03Z","timestamp":1754714703000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10836911\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,7]]},"references-count":33,"journal-issue":{"issue":"4"},"URL":"https:\/\/doi.org\/10.1109\/tsusc.2025.3528105","relation":{},"ISSN":["2377-3782","2377-3790"],"issn-type":[{"value":"2377-3782","type":"electronic"},{"value":"2377-3790","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,7]]}}}