{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,2]],"date-time":"2026-04-02T09:41:14Z","timestamp":1775122874373,"version":"3.50.1"},"reference-count":28,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"8","license":[{"start":{"date-parts":[[2024,8,1]],"date-time":"2024-08-01T00:00:00Z","timestamp":1722470400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2024,8,1]],"date-time":"2024-08-01T00:00:00Z","timestamp":1722470400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,8,1]],"date-time":"2024-08-01T00:00:00Z","timestamp":1722470400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"Grant from the Research Grants Council of the Hong Kong SAR","award":["CUHK14210923"],"award-info":[{"award-number":["CUHK14210923"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Comput.-Aided Des. Integr. Circuits Syst."],"published-print":{"date-parts":[[2024,8]]},"DOI":"10.1109\/tcad.2024.3368970","type":"journal-article","created":{"date-parts":[[2024,2,22]],"date-time":"2024-02-22T19:39:14Z","timestamp":1708630754000},"page":"2426-2439","source":"Crossref","is-referenced-by-count":3,"title":["Parmesan: Efficient Partitioning and Mapping Flow for DNN Training on General Device Topology"],"prefix":"10.1109","volume":"43","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8451-8100","authenticated-orcid":false,"given":"Lixin","family":"Liu","sequence":"first","affiliation":[{"name":"Department of Computer Science and Engineering, The Chinese University of Hong Kong, New Territories, Hong Kong"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-2204-280X","authenticated-orcid":false,"given":"Tianji","family":"Liu","sequence":"additional","affiliation":[{"name":"Department of Computer Science and Engineering, The Chinese University of Hong Kong, New Territories, Hong Kong"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8163-3114","authenticated-orcid":false,"given":"Bentian","family":"Jiang","sequence":"additional","affiliation":[{"name":"Department of Computer Science and Engineering, The Chinese University of Hong Kong, New Territories, Hong Kong"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0623-1590","authenticated-orcid":false,"given":"Evangeline F. Y.","family":"Young","sequence":"additional","affiliation":[{"name":"Department of Computer Science and Engineering, The Chinese University of Hong Kong, New Territories, Hong Kong"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref2","article-title":"BERT: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2018","journal-title":"arXiv:1810.04805"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"ref4","first-page":"463","article-title":"A unified architecture for accelerating distributed DNN training in heterogeneous GPU\/CPU clusters","volume-title":"Proc. OSDI","author":"Jiang"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3406703"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1145\/3492321.3519584"},{"key":"ref7","article-title":"Accurate, large minibatch SGD: Training ImageNet in 1 hour","author":"Goyal","year":"2017","journal-title":"arXiv:1706.02677"},{"key":"ref8","article-title":"One weird trick for parallelizing convolutional neural networks","author":"Krizhevsky","year":"2014","journal-title":"arXiv:1404.5997"},{"key":"ref9","first-page":"1","article-title":"Parallelized stochastic gradient descent","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Zinkevich"},{"key":"ref10","first-page":"1","article-title":"Project Adam: Building an efficient and scalable deep learning training system","volume-title":"Proc. OSDI","author":"Chilimbi"},{"key":"ref11","first-page":"1","article-title":"Beyond data and model parallelism for deep neural networks","volume-title":"Proc. MLSys","author":"Jia"},{"key":"ref12","article-title":"Megatron-LM: Training multi-billion parameter language models using model parallelism","author":"Shoeybi","year":"2019","journal-title":"arXiv:1909.08053"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.48550\/arxiv.1811.06965"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476145"},{"key":"ref15","first-page":"1","article-title":"TeraPipe: Token-level pipeline parallelism for training large-scale language models","volume-title":"Proc. ICML","author":"Li"},{"key":"ref16","first-page":"1","article-title":"Memory-efficient pipeline-parallel DNN training","volume-title":"Proc. ICML","author":"Narayanan"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1145\/3437801.3441593"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1145\/3341301.3359646"},{"key":"ref19","first-page":"1","article-title":"Piper: Multidimensional planner for DNN parallelization","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Tarnawski"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1016\/j.jpdc.2008.09.002"},{"key":"ref21","first-page":"1","article-title":"Device placement optimization with reinforcement learning","volume-title":"Proc. ICML","author":"Mirhoseini"},{"key":"ref22","first-page":"1676","article-title":"Spotlight: Optimizing device placement for training deep neural networks","volume-title":"Proc. ICML","author":"Gao"},{"key":"ref23","first-page":"1","article-title":"Reinforced genetic algorithm learning for optimizing computation graphs","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Paliwal"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS49936.2021.00109"},{"key":"ref25","first-page":"1","article-title":"Alpa: Automating inter- and intra-operator parallelism for distributed deep learning","volume-title":"Proc. OSDI","author":"Zheng"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1145\/140901.140903"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1145\/2966884.2966919"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00656"}],"container-title":["IEEE Transactions on Computer-Aided Design of Integrated Circuits and Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/43\/10604458\/10443652.pdf?arnumber=10443652","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,12]],"date-time":"2024-12-12T19:15:14Z","timestamp":1734030914000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10443652\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,8]]},"references-count":28,"journal-issue":{"issue":"8"},"URL":"https:\/\/doi.org\/10.1109\/tcad.2024.3368970","relation":{},"ISSN":["0278-0070","1937-4151"],"issn-type":[{"value":"0278-0070","type":"print"},{"value":"1937-4151","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,8]]}}}