{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,14]],"date-time":"2025-10-14T07:14:14Z","timestamp":1760426054489,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":59,"publisher":"ACM","license":[{"start":{"date-parts":[[2021,11,13]],"date-time":"2021-11-13T00:00:00Z","timestamp":1636761600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021,11,14]]},"DOI":"10.1145\/3458817.3476152","type":"proceedings-article","created":{"date-parts":[[2021,11,29]],"date-time":"2021-11-29T23:28:59Z","timestamp":1638228539000},"page":"1-14","source":"Crossref","is-referenced-by-count":9,"title":["KAISA"],"prefix":"10.1145","author":[{"given":"J. Gregory","family":"Pauloski","sequence":"first","affiliation":[{"name":"University of Chicago"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qi","family":"Huang","sequence":"additional","affiliation":[{"name":"University of Texas at Austin"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lei","family":"Huang","sequence":"additional","affiliation":[{"name":"Texas Advanced Computing Center"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shivaram","family":"Venkataraman","sequence":"additional","affiliation":[{"name":"University of Wisconsin, Madison"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kyle","family":"Chard","sequence":"additional","affiliation":[{"name":"University of Chicago"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ian","family":"Foster","sequence":"additional","affiliation":[{"name":"University of Chicago"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhao","family":"Zhang","sequence":"additional","affiliation":[{"name":"Texas Advanced Computing Center"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2021,11,13]]},"reference":[{"unstructured":"BERT. https:\/\/github.com\/google-research\/bert.  BERT. https:\/\/github.com\/google-research\/bert.","key":"e_1_3_2_1_1_1"},{"unstructured":"brain-segmentation-pytorch. https:\/\/www.kaggle.com\/mateuszbuda\/brain-segmentation-pytorch https:\/\/github.com\/mateuszbuda\/brain-segmentation-pytorch.  brain-segmentation-pytorch. https:\/\/www.kaggle.com\/mateuszbuda\/brain-segmentation-pytorch https:\/\/github.com\/mateuszbuda\/brain-segmentation-pytorch.","key":"e_1_3_2_1_2_1"},{"unstructured":"LGG segmentation dataset. https:\/\/www.kaggle.com\/mateuszbuda\/lgg-mri-segmentation.  LGG segmentation dataset. https:\/\/www.kaggle.com\/mateuszbuda\/lgg-mri-segmentation.","key":"e_1_3_2_1_3_1"},{"unstructured":"MLPerf. https:\/\/www.mlperf.org\/.  MLPerf. https:\/\/www.mlperf.org\/.","key":"e_1_3_2_1_4_1"},{"unstructured":"NVIDIA deep learning examples. https:\/\/github.com\/NVIDIA\/DeepLearningExamples.  NVIDIA deep learning examples. https:\/\/github.com\/NVIDIA\/DeepLearningExamples.","key":"e_1_3_2_1_5_1"},{"volume-title":"12th USENIX Symposium on Operating Systems Design and Implementation","year":"2016","author":"Abadi Mart\u00edn","key":"e_1_3_2_1_6_1"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_7_1","DOI":"10.1145\/3212734.3212763"},{"volume-title":"ICLR","year":"2017","author":"Ba Jimmy","key":"e_1_3_2_1_8_1"},{"volume-title":"Eigenvalue corrected noisy natural gradient. CoRR, abs\/1811.12565","year":"2018","author":"Bae Juhan","key":"e_1_3_2_1_9_1"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_10_1","DOI":"10.1137\/16M1080173"},{"volume-title":"Language models are few-shot learners. arXiv preprint arXiv:2005.14165","year":"2020","author":"Brown Tom B","key":"e_1_3_2_1_11_1"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_12_1","DOI":"10.1093\/imamat\/6.1.76"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_13_1","DOI":"10.1109\/IPDPS.2019.00038"},{"volume-title":"MXNet: A flexible and efficient machine learning library for heterogeneous distributed systems. arXiv preprint arXiv:1512.01274","year":"2015","author":"Chen Tianqi","key":"e_1_3_2_1_14_1"},{"key":"e_1_3_2_1_15_1","first-page":"613","volume-title":"14th USENIX Symposium on Networked Systems Design and Implementation (NSDI 17)","author":"Crankshaw Daniel","year":"2017"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_16_1","DOI":"10.1007\/s00211-007-0114-x"},{"volume-title":"BERT: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805","year":"2018","author":"Devlin Jacob","key":"e_1_3_2_1_17_1"},{"volume-title":"Advances in Neural Information Processing Systems","year":"2018","author":"George Thomas","key":"e_1_3_2_1_18_1"},{"key":"e_1_3_2_1_19_1","first-page":"2386","volume-title":"Advances in Neural Information Processing Systems","volume":"33","author":"Goldfarb Donald","year":"2020"},{"volume-title":"33rd International Conference on International Conference on Machine Learning","author":"Grosse Roger","first-page":"573","key":"e_1_3_2_1_20_1"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_21_1","DOI":"10.1109\/ICCV.2017.322"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_22_1","DOI":"10.1109\/CVPR.2016.90"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_23_1","DOI":"10.1145\/3373376.3378530"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_24_1","DOI":"10.1109\/CVPR.2017.243"},{"key":"e_1_3_2_1_25_1","first-page":"103","volume-title":"Advances in Neural Information Processing Systems","author":"Huang Yanping","year":"2019"},{"key":"e_1_3_2_1_26_1","first-page":"497","article-title":"Breaking the memory wall with optimal tensor rematerialization","volume":"2","author":"Jain Paras","year":"2020","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_27_1","first-page":"463","volume-title":"14th USENIX Symposium on Operating Systems Design and Implementation","author":"Jiang Yimin","year":"2020"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_28_1","DOI":"10.5555\/1051910"},{"key":"e_1_3_2_1_29_1","first-page":"1097","volume-title":"Advances in Neural Information Processing Systems","author":"Krizhevsky Alex","year":"2012"},{"volume-title":"Scale MLPerf-0.6 models on Google TPU-v3 Pods. arXiv preprint arXiv:1909.09756","year":"2019","author":"Kumar Sameer","key":"e_1_3_2_1_30_1"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_31_1","DOI":"10.1145\/2640087.2644155"},{"volume-title":"Microsoft COCO: Common objects in context","year":"2015","author":"Lin Tsung-Yi","key":"e_1_3_2_1_32_1"},{"volume-title":"Deep gradient compression: Reducing the communication bandwidth for distributed training. arXiv preprint arXiv:1712.01887","year":"2017","author":"Lin Yujun","key":"e_1_3_2_1_33_1"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_34_1","DOI":"10.5555\/3112655.3112866"},{"volume-title":"International Conference on Learning Representations","year":"2018","author":"Martens James","key":"e_1_3_2_1_35_1"},{"key":"e_1_3_2_1_36_1","first-page":"2408","volume-title":"International conference on machine learning","author":"Martens James","year":"2015"},{"volume-title":"Mixed precision training","year":"2018","author":"Micikevicius Paulius","key":"e_1_3_2_1_37_1"},{"key":"e_1_3_2_1_38_1","first-page":"561","volume-title":"13th USENIX Symposium on Operating Systems Design and Implementation","author":"Moritz Philipp","year":"2018"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_39_1","DOI":"10.1145\/3341301.3359646"},{"volume-title":"Heterogeneity-aware cluster scheduling policies for deep learning workloads. arXiv preprint arXiv:2008.09213","year":"2020","author":"Narayanan Deepak","key":"e_1_3_2_1_40_1"},{"volume-title":"Flexible, high-performance ml serving. arXiv preprint arXiv:1712.06139","year":"2017","author":"Olston Christopher","key":"e_1_3_2_1_41_1"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_42_1","DOI":"10.1109\/CVPR.2019.01264"},{"key":"e_1_3_2_1_43_1","first-page":"8024","volume-title":"Advances in Neural Information Processing Systems 32","author":"Paszke Adam","year":"2019"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_44_1","DOI":"10.5555\/3433701.3433826"},{"key":"e_1_3_2_1_45_1","first-page":"693","volume-title":"Advances in Neural Information Processing Systems","author":"Recht Benjamin","year":"2011"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_46_1","DOI":"10.1007\/978-3-319-24574-4_28"},{"volume-title":"Horovod: Fast and easy distributed deep learning in TensorFlow. arXiv preprint arXiv:1802.05799","year":"2018","author":"Sergeev Alexander","key":"e_1_3_2_1_47_1"},{"key":"e_1_3_2_1_48_1","first-page":"90","volume-title":"Communication-optimal parallel 2.5d matrix multiplication and lu factorization algorithms","author":"Solomonik Edgar","year":"2011"},{"volume-title":"Proceedings of the 48th International Conference on Parallel Processing: Workshops, ICPP 2019","year":"2019","author":"Tsuji Yohei","key":"e_1_3_2_1_49_1"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_50_1","DOI":"10.1145\/3394486.3403265"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_51_1","DOI":"10.1145\/3178487.3178491"},{"unstructured":"Wikipedia. Wikipedia Corpus. https:\/\/meta.wikimedia.org\/wiki\/Data_dump_torrents#English_Wikipedia.  Wikipedia. Wikipedia Corpus. https:\/\/meta.wikimedia.org\/wiki\/Data_dump_torrents#English_Wikipedia.","key":"e_1_3_2_1_52_1"},{"volume-title":"Google's neural machine translation system: Bridging the gap between human and machine translation. arXiv preprint arXiv:1609.08144","year":"2016","author":"Wu Yonghui","key":"e_1_3_2_1_53_1"},{"key":"e_1_3_2_1_54_1","first-page":"595","volume-title":"13th USENIX Symposium on Operating Systems Design and Implementation (OSDI 18)","author":"Xiao Wencong","year":"2018"},{"key":"e_1_3_2_1_55_1","first-page":"533","volume-title":"14th USENIX Symposium on Operating Systems Design and Implementation","author":"Xiao Wencong","year":"2020"},{"volume-title":"Large batch optimization for deep learning: Training BERT in 76 minutes. arXiv preprint arXiv:1904.00962","year":"2019","author":"You Yang","key":"e_1_3_2_1_56_1"},{"key":"e_1_3_2_1_57_1","first-page":"8082","volume-title":"Advances in Neural Information Processing Systems","author":"Zhang Guodong","year":"2019"},{"key":"e_1_3_2_1_58_1","series-title":"Proceedings of Machine Learning Research","first-page":"5852","volume-title":"Jennifer Dy and Andreas Krause","author":"Zhang Guodong","year":"2018"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_59_1","DOI":"10.1109\/ICCV.2015.11"}],"event":{"sponsor":["SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing","IEEE CS"],"acronym":"SC '21","name":"SC '21: The International Conference for High Performance Computing, Networking, Storage and Analysis","location":"St. Louis Missouri"},"container-title":["Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3458817.3476152","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3458817.3476152","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T17:49:06Z","timestamp":1750268946000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3458817.3476152"}},"subtitle":["an adaptive second-order optimizer framework for deep neural networks"],"short-title":[],"issued":{"date-parts":[[2021,11,13]]},"references-count":59,"alternative-id":["10.1145\/3458817.3476152","10.1145\/3458817"],"URL":"https:\/\/doi.org\/10.1145\/3458817.3476152","relation":{},"subject":[],"published":{"date-parts":[[2021,11,13]]}}}