{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,24]],"date-time":"2025-11-24T19:58:10Z","timestamp":1764014290486,"version":"3.45.0"},"reference-count":40,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"1","license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"am","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"NSF CAREER","award":["2146421"],"award-info":[{"award-number":["2146421"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Comput.-Aided Des. Integr. Circuits Syst."],"published-print":{"date-parts":[[2025,1]]},"DOI":"10.1109\/tcad.2024.3404413","type":"journal-article","created":{"date-parts":[[2024,5,22]],"date-time":"2024-05-22T13:39:37Z","timestamp":1716385177000},"page":"266-279","source":"Crossref","is-referenced-by-count":0,"title":["Rethinking Latency-Aware DNN Design With GPU Tail Effect Analysis"],"prefix":"10.1109","volume":"44","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-4880-6658","authenticated-orcid":false,"given":"Fuxun","family":"Yu","sequence":"first","affiliation":[{"name":"Department of Research and Development, Microsoft Corporation, Redmond, WA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3556-9358","authenticated-orcid":false,"given":"Zirui","family":"Xu","sequence":"additional","affiliation":[{"name":"Department of Research and Development, CVS Health Corporation, Woonsocket, RI, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1153-7087","authenticated-orcid":false,"given":"Longfei","family":"Shangguan","sequence":"additional","affiliation":[{"name":"Department of Computer Science, University of Pittsburgh, Pittsburgh, PA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8003-9738","authenticated-orcid":false,"given":"Di","family":"Wang","sequence":"additional","affiliation":[{"name":"Department of Research and Development, Microsoft Corporation, Redmond, WA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dimitrios","family":"Stamoulis","sequence":"additional","affiliation":[{"name":"Department of Research and Development, Microsoft Corporation, Redmond, WA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8124-5744","authenticated-orcid":false,"given":"Rishi","family":"Madhok","sequence":"additional","affiliation":[{"name":"Department of Research and Development, Microsoft Corporation, Redmond, WA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Nikolaos","family":"Karianakis","sequence":"additional","affiliation":[{"name":"Department of Research and Development, Microsoft Corporation, Redmond, WA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ang","family":"Li","sequence":"additional","affiliation":[{"name":"Department of Electrical and Computer Engineering, University of Maryland at College Park, College Park, MD, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"ChenChen","family":"Liu","sequence":"additional","affiliation":[{"name":"Department of Computer Science and Electrical Engineering, University of Maryland at Baltimore County, Baltimore, MD, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1486-8412","authenticated-orcid":false,"given":"Yiran","family":"Chen","sequence":"additional","affiliation":[{"name":"Department of Electrical and Computer Engineering, Duke University, Durham, NC, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2790-976X","authenticated-orcid":false,"given":"Xiang","family":"Chen","sequence":"additional","affiliation":[{"name":"Department of Electrical and Computer Engineering, George Mason University, Fairfax, VA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","first-page":"578","article-title":"{TVM}: An automated end-to-end optimizing compiler for deep learning","volume-title":"Proc. 13th {USENIX} Symp. Oper. Syst. Design Implement. ({OSDI})","author":"Chen"},{"doi-asserted-by":"publisher","key":"ref2","DOI":"10.1109\/CVPR.2019.01166"},{"doi-asserted-by":"publisher","key":"ref3","DOI":"10.1109\/CVPR.2009.5206848"},{"doi-asserted-by":"publisher","key":"ref4","DOI":"10.48550\/arXiv.2010.11929"},{"doi-asserted-by":"publisher","key":"ref5","DOI":"10.1109\/CVPR52729.2023.01544"},{"doi-asserted-by":"publisher","key":"ref6","DOI":"10.1109\/ICASSP.2013.6638947"},{"doi-asserted-by":"publisher","key":"ref7","DOI":"10.1109\/CVPR.2016.90"},{"doi-asserted-by":"publisher","key":"ref8","DOI":"10.24963\/ijcai.2018\/309"},{"doi-asserted-by":"publisher","key":"ref9","DOI":"10.1109\/CVPR.2019.00447"},{"doi-asserted-by":"publisher","key":"ref10","DOI":"10.1007\/978-3-030-01234-2_48"},{"doi-asserted-by":"publisher","key":"ref11","DOI":"10.1109\/ICCV.2017.155"},{"key":"ref12","article-title":"MobileNets: Efficient convolutional neural networks for mobile vision applications","author":"Howard","year":"2017","journal-title":"arXiv:1704.04861"},{"doi-asserted-by":"publisher","key":"ref13","DOI":"10.1145\/3341301.3359630"},{"doi-asserted-by":"publisher","key":"ref14","DOI":"10.1109\/CVPR42600.2020.00160"},{"year":"2020","author":"Lin","article-title":"Hrankplus github repo.","key":"ref15"},{"doi-asserted-by":"publisher","key":"ref16","DOI":"10.1007\/978-3-319-10602-1_48"},{"doi-asserted-by":"publisher","key":"ref17","DOI":"10.1109\/ICCV48922.2021.00986"},{"doi-asserted-by":"publisher","key":"ref18","DOI":"10.1109\/ICCV.2017.298"},{"year":"2020","article-title":"Deep learning inference service at Microsoft","key":"ref19"},{"volume-title":"Cuda pro tips (page 13).","year":"2020","key":"ref20"},{"volume-title":"Cudnn kernel invoking logic.","year":"2020","key":"ref21"},{"volume-title":"DNN compiling stack and kernel base.","year":"2020","key":"ref22"},{"volume-title":"Jetson devices.","year":"2020","key":"ref23"},{"volume-title":"Nsight compute | Nvidia.","year":"2020","key":"ref24"},{"volume-title":"Nvidia CUDNN documentation | kernel selection heuristics.","year":"2020","key":"ref25"},{"volume-title":"Nvidia multi instance GPU (MIG).","year":"2020","key":"ref26"},{"volume-title":"Nvidia multi process service (MPS).","year":"2020","key":"ref27"},{"volume-title":"Nvidia Titan v | volta architecture.","year":"2020","key":"ref28"},{"key":"ref29","article-title":"A survey on efficient vision transformers: Algorithms, techniques, and performance benchmarking","author":"Papa","year":"2023","journal-title":"arXiv:2309.02031"},{"doi-asserted-by":"publisher","key":"ref30","DOI":"10.1145\/2499370.2462176"},{"doi-asserted-by":"publisher","key":"ref31","DOI":"10.1109\/CVPR.2018.00474"},{"key":"ref32","article-title":"Very deep convolutional networks for large-scale image recognition","author":"Simonyan","year":"2014","journal-title":"arXiv:1409.1556"},{"doi-asserted-by":"publisher","key":"ref33","DOI":"10.1109\/CVPR.2019.00293"},{"key":"ref34","article-title":"EfficientNet: Rethinking model scaling for convolutional neural networks","author":"Tan","year":"2019","journal-title":"arXiv:1905.11946"},{"year":"2016","author":"Vanholder","article-title":"Efficient inference with TensorRT.","key":"ref35"},{"doi-asserted-by":"publisher","key":"ref36","DOI":"10.1109\/CVPR.2019.01099"},{"doi-asserted-by":"publisher","key":"ref37","DOI":"10.1109\/CVPR52729.2023.01779"},{"doi-asserted-by":"publisher","key":"ref38","DOI":"10.1007\/978-3-030-01249-6_18"},{"key":"ref39","article-title":"Salus: Fine-grained GPU sharing primitives for deep learning applications","author":"Yu","year":"2019","journal-title":"arXiv:1902.04610"},{"doi-asserted-by":"publisher","key":"ref40","DOI":"10.1145\/3458864.3467882"}],"container-title":["IEEE Transactions on Computer-Aided Design of Integrated Circuits and Systems"],"original-title":[],"link":[{"URL":"https:\/\/ieeexplore.ieee.org\/ielam\/43\/10814105\/10537049-aam.pdf","content-type":"application\/pdf","content-version":"am","intended-application":"syndication"},{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/43\/10814105\/10537049.pdf?arnumber=10537049","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,24]],"date-time":"2025-11-24T19:00:03Z","timestamp":1764010803000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10537049\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,1]]},"references-count":40,"journal-issue":{"issue":"1"},"URL":"https:\/\/doi.org\/10.1109\/tcad.2024.3404413","relation":{},"ISSN":["0278-0070","1937-4151"],"issn-type":[{"type":"print","value":"0278-0070"},{"type":"electronic","value":"1937-4151"}],"subject":[],"published":{"date-parts":[[2025,1]]}}}