{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T01:13:53Z","timestamp":1740100433400,"version":"3.37.3"},"reference-count":49,"publisher":"IEEE","license":[{"start":{"date-parts":[[2021,11,1]],"date-time":"2021-11-01T00:00:00Z","timestamp":1635724800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2021,11,1]],"date-time":"2021-11-01T00:00:00Z","timestamp":1635724800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100000001","name":"NSF","doi-asserted-by":"publisher","award":["#1937500"],"award-info":[{"award-number":["#1937500"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021,11,1]]},"DOI":"10.1109\/iccad51958.2021.9643503","type":"proceedings-article","created":{"date-parts":[[2021,12,23]],"date-time":"2021-12-23T23:06:46Z","timestamp":1640300806000},"page":"1-9","source":"Crossref","is-referenced-by-count":1,"title":["ScaleDNN: Data Movement Aware DNN Training on Multi-GPU"],"prefix":"10.1109","author":[{"given":"Weizheng","family":"Xu","sequence":"first","affiliation":[{"name":"University of Pittsburgh"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ashutosh","family":"Pattnaik","sequence":"additional","affiliation":[{"name":"Penn State University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Geng","family":"Yuan","sequence":"additional","affiliation":[{"name":"Northeastern University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yanzhi","family":"Wang","sequence":"additional","affiliation":[{"name":"Northeastern University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Youtao","family":"Zhang","sequence":"additional","affiliation":[{"name":"University of Pittsburgh"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xulong","family":"Tang","sequence":"additional","affiliation":[{"name":"University of Pittsburgh"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.2017.2761740"},{"journal-title":"Horovod fast and easy distributed deep learning in tensorflow","year":"2018","author":"sergeev","key":"ref38"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2019.2935967"},{"journal-title":"Nsight Systems User Guide","year":"2021","key":"ref32"},{"journal-title":"Profiler User's Guide","year":"2019","key":"ref31"},{"journal-title":"NVIDIA Collective Communication Library (NCCL) Documentation","year":"2019","key":"ref30"},{"key":"ref37","article-title":"Weightless: Lossy weight encoding for deep neural network compression","author":"reagan","year":"2018","journal-title":"International Conference on Machine Learning"},{"journal-title":"Know What You Don't Know Unanswerable Questions for SQuAD","year":"2018","author":"rajpurkar","key":"ref36"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/SC41405.2020.00024"},{"key":"ref34","article-title":"Pytorch: An imperative style, highperformance deep learning library","author":"paszke","year":"2019","journal-title":"NIPS"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1145\/3341301.3359646"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/EPEPS.2010.5642789"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-63393-6_8"},{"key":"ref2","article-title":"Adacomp: Adaptive residual gradient compression for data-parallel distributed training","author":"chen","year":"2018","journal-title":"AAAI"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1145\/3126908.3126933"},{"key":"ref20","article-title":"Tiny imagenet visual recognition challenge","author":"le","year":"2015","journal-title":"CS 231N"},{"key":"ref22","article-title":"Evaluating modern gpu interconnect: Pcie, nvlink, nv-sli, nvswitch and gpudirect","author":"li","year":"2019","journal-title":"IEEE TPDS"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/DAC18072.2020.9218538"},{"key":"ref24","article-title":"Scaling distributed machine learning with the parameter server","author":"li","year":"2014","journal-title":"OSDI"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1145\/2741948.2741965"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1145\/3363554"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1145\/3373376.3378499"},{"journal-title":"Deep compression Compressing deep neural networks with pruning trained quantization and huffman coding","year":"2015","author":"han","key":"ref10"},{"journal-title":"Distilling the knowledge in a neural network","year":"2015","author":"hinton","key":"ref11"},{"key":"ref40","article-title":"Deep neural networks for object detection","author":"szegedy","year":"2013","journal-title":"NIPS"},{"journal-title":"DenseNet implementing efficient convnet descriptor pyramids","year":"2014","author":"iandola","key":"ref12"},{"key":"ref13","article-title":"Beyond data and model parallelism for deep neural networks","author":"jia","year":"2019","journal-title":"SysML"},{"key":"ref14","article-title":"A Unified Architecture for Accelerating Distributed DNN Training in Heterogeneous GPU\/CPU Clusters","author":"jiang","year":"2020","journal-title":"OSDI"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/TCAD.2020.3013194"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00065"},{"journal-title":"A study of BFLOAT16 for Deep Learning Training","year":"2019","author":"kalamkar","key":"ref17"},{"key":"ref18","article-title":"Multi-gpu system design with memory networks","author":"kim","year":"2014","journal-title":"Micro"},{"journal-title":"One weird trick for parallelizing convolutional neural networks","year":"2014","author":"krizhevsky","key":"ref19"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1145\/2901318.2901323"},{"journal-title":"Towards the Limit of Network Quantization","year":"2016","author":"choi","key":"ref3"},{"journal-title":"BERT Pre-training of deep bidirectional transformers for language understanding","year":"2018","author":"devlin","key":"ref6"},{"journal-title":"Pre-dicting parameters in deep learning","year":"2013","author":"denil","key":"ref5"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1145\/3240765.3243484"},{"journal-title":"Accurate large minibatch sgd Training imagenet in 1 hour","year":"2017","author":"goyal","key":"ref7"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33015676"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1145\/3195970.3199847"},{"key":"ref46","first-page":"172","article-title":"Blink: Fast and generic collectives for distributed ml","volume":"2","author":"wang","year":"2020","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00821"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1145\/3442442.3452055"},{"key":"ref47","article-title":"Bfloat16: the secret to high performance on cloud tpus","author":"wang","year":"2019","journal-title":"Google Cloud"},{"key":"ref42","article-title":"Com-puting with near data","author":"tang","year":"2019","journal-title":"Proceedings of the 2019 ACM SIGMETRICS International Conference on Measurement and Modeling of Computer Systems (SIGMETRICS)"},{"key":"ref41","article-title":"Evaluating on-node gpu interconnects for deep learning workloads","author":"tallent","year":"2017","journal-title":"International Workshop on Performance Modeling Benchmarking and Simulation of High Performance Computer Systems"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1145\/3180155.3180220"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1145\/3123939.3123954"}],"event":{"name":"2021 IEEE\/ACM International Conference On Computer Aided Design (ICCAD)","start":{"date-parts":[[2021,11,1]]},"location":"Munich, Germany","end":{"date-parts":[[2021,11,4]]}},"container-title":["2021 IEEE\/ACM International Conference On Computer Aided Design (ICCAD)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9643423\/9643432\/09643503.pdf?arnumber=9643503","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,8,3]],"date-time":"2022-08-03T00:12:13Z","timestamp":1659485533000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9643503\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,11,1]]},"references-count":49,"URL":"https:\/\/doi.org\/10.1109\/iccad51958.2021.9643503","relation":{},"subject":[],"published":{"date-parts":[[2021,11,1]]}}}