{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T04:08:38Z","timestamp":1750219718920,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":92,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,8,10]],"date-time":"2023-08-10T00:00:00Z","timestamp":1691625600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,8,10]]},"DOI":"10.1145\/3589013.3596677","type":"proceedings-article","created":{"date-parts":[[2023,8,11]],"date-time":"2023-08-11T07:53:54Z","timestamp":1691740434000},"page":"33-40","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["AutoConstruct: Automated Neural Surrogate Model Building and Deployment for HPC Applications"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8376-6647","authenticated-orcid":false,"given":"Wenqian","family":"Dong","sequence":"first","affiliation":[{"name":"Florida International University, Miami, FL, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5541-433X","authenticated-orcid":false,"given":"Jie","family":"Ren","sequence":"additional","affiliation":[{"name":"College of William and Mary, Williamsburg, VA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,8,11]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Tensorflow: A system for large-scale machine learning. In 12th {USENIX} symposium on operating systems design and implementation ({OSDI} 16). 265--283.","author":"Abadi Mart\u00edn","year":"2016","unstructured":"Mart\u00edn Abadi, Paul Barham, Jianmin Chen, Zhifeng Chen, Andy Davis, Jeffrey Dean, Matthieu Devin, Sanjay Ghemawat, Geoffrey Irving, Michael Isard, et al. 2016. Tensorflow: A system for large-scale machine learning. In 12th {USENIX} symposium on operating systems design and implementation ({OSDI} 16). 265--283."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/1806596.1806620"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3295500.3356202"},{"volume-title":"Benchmarking modern multiprocessors","author":"Bienia Christian","key":"e_1_3_2_1_4_1","unstructured":"Christian Bienia. 2011. Benchmarking modern multiprocessors. Princeton University."},{"volume-title":"Building Machine Learning and Deep Learning Models on Google Cloud Platform","author":"Bisong Ekaba","key":"e_1_3_2_1_5_1","unstructured":"Ekaba Bisong. 2019. Google AutoML: Cloud Vision. In Building Machine Learning and Deep Learning Models on Google Cloud Platform. Springer, 581--598."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-34356-9_41"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.5555\/1712666.1712670"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CGO.2015.7054203"},{"key":"e_1_3_2_1_9_1","volume-title":"Solving the quantum many-body problem with artificial neural networks. Science 355, 6325","author":"Carleo Giuseppe","year":"2017","unstructured":"Giuseppe Carleo and Matthias Troyer. 2017. Solving the quantum many-body problem with artificial neural networks. Science 355, 6325 (2017), 602--606."},{"key":"e_1_3_2_1_10_1","volume-title":"Training deep nets with sublinear memory cost. arXiv preprint arXiv:1604.06174","author":"Chen Tianqi","year":"2016","unstructured":"Tianqi Chen, Bing Xu, Chiyuan Zhang, and Carlos Guestrin. 2016. Training deep nets with sublinear memory cost. arXiv preprint arXiv:1604.06174 (2016)."},{"key":"e_1_3_2_1_11_1","volume-title":"cudnn: Efficient primitives for deep learning. arXiv preprint arXiv:1410.0759","author":"Chetlur Sharan","year":"2014","unstructured":"Sharan Chetlur, Cliff Woolley, Philippe Vandermersch, Jonathan Cohen, John Tran, Bryan Catanzaro, and Evan Shelhamer. 2014. cudnn: Efficient primitives for deep learning. arXiv preprint arXiv:1410.0759 (2014)."},{"volume-title":"Design Automation Conference.","author":"Chippa V. K.","key":"e_1_3_2_1_12_1","unstructured":"V. K. Chippa, D. Mohapatra, A. Raghunathan, K. Roy, and S. T. Chakradhar. 2010. Scalable effort hardware design: Exploiting algorithmic resilience for energy efficiency. In Design Automation Conference."},{"volume-title":"Data Analysis for Direct Numerical Simulations of Turbulent Combustion","author":"Domingo Pascale","key":"e_1_3_2_1_13_1","unstructured":"Pascale Domingo, Zacharias Nikolaou, Andr\u00e9a Seltz, and Luc Vervisch. 2020. From Discrete and Iterative Deconvolution Operators to Machine Learning for Premixed Turbulent Combustion Modeling. In Data Analysis for Direct Numerical Simulations of Turbulent Combustion. Springer, 215--232."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3295500.3356147"},{"key":"e_1_3_2_1_15_1","volume-title":"Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis","author":"Dong Wenqian","year":"2020","unstructured":"Wenqian Dong, Zhen Xie, Gokcen Kestor, and Dong Li. 2020. Smart-PGSim: Using Neural Network to Accelerate AC-OPF Power Grid Simulation. In Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis (Atlanta, Georgia) (SC '20). IEEE Press, Article 63, 15 pages."},{"key":"e_1_3_2_1_16_1","volume-title":"Constrained physicsinformed deep learning for stable system identification and control of unknown linear systems. arXiv e-prints","author":"Drgona Jan","year":"2020","unstructured":"Jan Drgona, Aaron Tuor, and Draguna Vrabie. 2020. Constrained physicsinformed deep learning for stable system identification and control of unknown linear systems. arXiv e-prints (2020), arXiv--2004."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"crossref","unstructured":"Hadi Esmaeilzadeh Adrian Sampson Luis Ceze and Doug Burger. 2012. Architecture Support for Disciplined Approximate Programming. In ASPLOS.","DOI":"10.1145\/2150976.2151008"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2012.48"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"crossref","unstructured":"Jianwei Feng and Dong Huang. 2021. Optimal Gradient Checkpoint Search for Arbitrary Computation Graphs. arXiv:1808.00079 [cs.LG]","DOI":"10.1109\/CVPR46437.2021.01127"},{"key":"e_1_3_2_1_20_1","volume-title":"Machine Learning Molecular Dynamics for the Simulation of Infrared Spectra . Chemical Science","author":"Gastegger Michael","year":"2017","unstructured":"Michael Gastegger, J\u00f6rg Behlerb, and Philipp Marquetand. 2017. Machine Learning Molecular Dynamics for the Simulation of Infrared Spectra . Chemical Science (2017), 6695--7270."},{"key":"e_1_3_2_1_21_1","volume-title":"Nguyen","author":"Goiri Inigo","year":"2015","unstructured":"Inigo Goiri, Ricardo Bianchini, Santosh Nagarakatte, and Thu D. Nguyen. 2015. ApproxHadoop: Bringing Approximations to MapReduce Frameworks. In ASPLOS."},{"volume-title":"Proceedings of the International Conference for High Performance Computing, Networking, Storage, and Analysis.","author":"Guo L.","key":"e_1_3_2_1_22_1","unstructured":"L. Guo, D. Li, I. Laguna, and M. Schulz. 2018. FlipTracker: Understanding Natural Error Resilience in HPC Applications. In Proceedings of the International Conference for High Performance Computing, Networking, Storage, and Analysis."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"crossref","unstructured":"J. Han and M. Orshansky. 2013. Approximate computing: An emerging paradigm for energy-efficient design. In ETS.","DOI":"10.1109\/ETS.2013.6569370"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/174662.174663"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/3373376.3378465"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"crossref","unstructured":"Mark Hildebrand Jawad Khan Sanjeev Trika Jason Lowe-Power and Venkatesh Akella. 2020. AutoTM: Automatic Tensor Movement in Heterogeneous Memory Systems Using Integer Linear Programming. In International Conference on Architectural Support for Programming Languages and Operating Systems.","DOI":"10.1145\/3373376.3378465"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"crossref","unstructured":"Henry Hoffmann Stelios Sidiroglou Michael Carbin Sasa Misailovic Anant Agarwal and Martin Rinard. 2011. Dynamic Knobs for Responsive Power-aware Computing. In ASPLOS.","DOI":"10.1145\/1950365.1950390"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3373376.3378530"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3373376.3378530"},{"key":"e_1_3_2_1_30_1","volume-title":"Proceedings of Machine Learning and Systems (MLSys).","author":"Jain Paras","year":"2020","unstructured":"Paras Jain, Ajay Jain, Aniruddha Nrusimha, Amir Gholami, Pieter Abbeel, Joseph Gonzalez, Kurt Keutzer, and Ion Stoica. 2020. Checkmate: Breaking the Memory Wall with Optimal Tensor Rematerialization. In Proceedings of Machine Learning and Systems (MLSys)."},{"key":"e_1_3_2_1_31_1","volume-title":"Meshfreeflownet: A physics-constrained deep continuous space-time super-resolution framework. arXiv preprint arXiv:2005.01463","author":"Jiang Chiyu Max","year":"2020","unstructured":"Chiyu Max Jiang, Soheil Esmaeilzadeh, Kamyar Azizzadenesheli, Karthik Kashinath, Mustafa Mustafa, Hamdi A Tchelepi, Philip Marcus, Anima Anandkumar, et al. 2020. Meshfreeflownet: A physics-constrained deep continuous space-time super-resolution framework. arXiv preprint arXiv:2005.01463 (2020)."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/3292500.3330648"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.compstruc.2006.02.015"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/2749469.2750371"},{"volume-title":"Computer Graphics Forum","author":"Kim Byungsoo","key":"e_1_3_2_1_35_1","unstructured":"Byungsoo Kim, Vinicius C Azevedo, Nils Thuerey, Theodore Kim, Markus Gross, and Barbara Solenthaler. 2019. Deep fluids: A generative network for parameterized fluid simulations. In Computer Graphics Forum, Vol. 38. Wiley Online Library, 59--70."},{"key":"e_1_3_2_1_36_1","unstructured":"Marisa Kirisame Steven Lyubomirsky Altan Haan Jennifer Brennan Mike He Jared Roesch Tianqi Chen and Zachary Tatlock. 2021. Dynamic Tensor Rematerialization. arXiv:2006.09616 [cs.LG]"},{"key":"e_1_3_2_1_37_1","volume-title":"Pramod Bhatotia, Christof Fetzer, and Rodrigo Rodrigues.","author":"Krishnan Dhanya R.","year":"2016","unstructured":"Dhanya R. Krishnan, Do Le Quoc, Pramod Bhatotia, Christof Fetzer, and Rodrigo Rodrigues. 2016. IncApprox: A Data Analytics System for Incremental Approximate Computing. In WWW."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/2908080.2908087"},{"key":"e_1_3_2_1_39_1","volume-title":"Fourier neural operator for parametric partial differential equations. arXiv preprint arXiv:2010.08895","author":"Li Zongyi","year":"2020","unstructured":"Zongyi Li, Nikola Kovachki, Kamyar Azizzadenesheli, Burigede Liu, Kaushik Bhattacharya, AndrewStuart, and Anima Anandkumar. 2020. Fourier neural operator for parametric partial differential equations. arXiv preprint arXiv:2010.08895 (2020)."},{"key":"e_1_3_2_1_40_1","volume-title":"TeraPipe: Token-Level Pipeline Parallelism for Training Large-Scale Language Models. In International Conference on Machine Learning.","author":"Li Zhuohan","year":"2021","unstructured":"Zhuohan Li, Siyuan Zhuang, Shiyuan Guo, Danyang Zhuo, Hao Zhang, Dawn Song, and Ion Stoica. 2021. TeraPipe: Token-Level Pipeline Parallelism for Training Large-Scale Language Models. In International Conference on Machine Learning."},{"key":"e_1_3_2_1_41_1","volume-title":"Neural network based constitutive model for elastomeric foams. Engineering structures 30, 7","author":"Liang Guanghui","year":"2008","unstructured":"Guanghui Liang and K Chandrashekhara. 2008. Neural network based constitutive model for elastomeric foams. Engineering structures 30, 7 (2008), 2002--2011."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1017\/jfm.2016.615"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.14778\/3476249.3476254"},{"key":"e_1_3_2_1_44_1","unstructured":"Yunjie Liu Evan Racah Joaquin Correa Amir Khosrowshahi David Lavers Kenneth Kunkel Michael Wehner William Collins et al. 2016. Application of deep convolutional neural networks for detecting extreme weather in climate datasets. arXiv preprint arXiv:1605.01156 (2016)."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.cpc.2020.107624"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1145\/3007787.3001144"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178320"},{"key":"e_1_3_2_1_48_1","volume-title":"Recurrent neural network architecture search for geophysical emulation. arXiv preprint arXiv:2004.10928","author":"Maulik Romit","year":"2020","unstructured":"Romit Maulik, Romain Egele, Bethany Lusch, and Prasanna Balaprakash. 2020. Recurrent neural network architecture search for geophysical emulation. arXiv preprint arXiv:2004.10928 (2020)."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/MCSE.2017.57"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"crossref","unstructured":"Sasa Misailovic Stelios Sidiroglou Henry Hoffmann and Martin Rinard. 2010. Quality of Service Profiling. In ICSE.","DOI":"10.1145\/1806799.1806808"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2015.7056066"},{"key":"e_1_3_2_1_52_1","volume-title":"Proceedings of the International Conference on Machine Learning (ICML).","author":"Narayanan Deepak","year":"2021","unstructured":"Deepak Narayanan, Amar Phanishayee, Kaiyu Shi, Xie Chen, and Matei Zaharia. 2021. Memory-Efficient Pipeline-Parallel DNN Training. In Proceedings of the International Conference on Machine Learning (ICML)."},{"key":"e_1_3_2_1_53_1","volume-title":"Dmitri Vainbrand, Prethvi Kashinkunti, Julie Bernauer, Bryan Catanzaro, Amar Phanishayee, and Matei Zaharia.","author":"Narayanan Deepak","year":"2021","unstructured":"Deepak Narayanan, Mohammad Shoeybi, Jared Casper, Patrick LeGresley, Mostofa Patwary, Vijay Anand Korthikanti, Dmitri Vainbrand, Prethvi Kashinkunti, Julie Bernauer, Bryan Catanzaro, Amar Phanishayee, and Matei Zaharia. 2021. Efficient Large-Scale Language Model Training on GPU Clusters Using Megatron-LM. arXiv:2104.04473 [cs.CL]"},{"key":"e_1_3_2_1_54_1","volume-title":"Gianni De Fabritiis, and Cecilia Clementi","author":"No\u00e9 Frank","year":"2020","unstructured":"Frank No\u00e9, Gianni De Fabritiis, and Cecilia Clementi. 2020. Machine learning for protein folding and dynamics. Current opinion in structural biology 60 (2020), 77--84."},{"key":"e_1_3_2_1_55_1","unstructured":"Tom O'Malley Elie Bursztein James Long Fran\u00e7ois Chollet Haifeng Jin Luca Invernizzi et al. 2019. Keras Tuner. Retrieved May 21 (2019) 2020."},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476216"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1145\/3373376.3378505"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1145\/3373376.3378505"},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1109\/WCRE.2002.1173062"},{"key":"e_1_3_2_1_60_1","volume-title":"Training Large Neural Networks with Constant Memory using a New Execution Algorithm. CoRR abs\/2002.05645","author":"Pudipeddi Bharadwaj","year":"2020","unstructured":"Bharadwaj Pudipeddi, Maral Mesmakhosroshahi, Jinwen Xi, and Sujeeth Bharadwaj. 2020. Training Large Neural Networks with Constant Memory using a New Execution Algorithm. CoRR abs\/2002.05645 (2020)."},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC41405.2020.00024"},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476205"},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476205"},{"key":"e_1_3_2_1_64_1","volume-title":"Sentinel: Efficient Tensor Migration and Allocation on Heterogeneous Memory Systems for Deep Learning. In International Symposium on High Performance Computer Architecture (HPCA).","author":"Ren Jie","year":"2020","unstructured":"Jie Ren, Jiaolin Luo, Kai Wu, Minjia Zhang, Hyeran Jeon, and Dong Li. 2020. Sentinel: Efficient Tensor Migration and Allocation on Heterogeneous Memory Systems for Deep Learning. In International Symposium on High Performance Computer Architecture (HPCA)."},{"key":"e_1_3_2_1_65_1","volume-title":"Sentinel: Efficient Tensor Migration and Allocation on Heterogeneous Memory Systems for Deep Learning. In International Symposium on High Performance Computer Architecture (HPCA).","author":"Ren Jie","year":"2021","unstructured":"Jie Ren, Jiaolin Luo, Kai Wu, Minjia Zhang, Hyeran Jeon, and Dong Li. 2021. Sentinel: Efficient Tensor Migration and Allocation on Heterogeneous Memory Systems for Deep Learning. In International Symposium on High Performance Computer Architecture (HPCA)."},{"key":"e_1_3_2_1_66_1","volume-title":"Olatunji Ruwase, Shuangyan Yang, Minjia Zhang, Dong Li, and Yuxiong He.","author":"Ren Jie","year":"2021","unstructured":"Jie Ren, Samyam Rajbhandari, Reza Yazdani Aminabadi, Olatunji Ruwase, Shuangyan Yang, Minjia Zhang, Dong Li, and Yuxiong He. 2021. ZeRO-Offload: Democratizing Billion-Scale Model Training. arXiv:2101.06840 [cs.DC]"},{"key":"e_1_3_2_1_67_1","volume-title":"ZeRO-Offload: Democratizing Billion-Scale Model Training. In USENIX Annual Technical Conference.","author":"Ren Jie","year":"2021","unstructured":"Jie Ren, Samyam Rajbhandari, Reza Yazdani Aminabadi, Olatunji Ruwase, Shuangyan Yang, Minjia Zhang, Dong Li, and Yuxiong He. 2021. ZeRO-Offload: Democratizing Billion-Scale Model Training. In USENIX Annual Technical Conference."},{"key":"e_1_3_2_1_68_1","volume-title":"Memoryefficient Neural Network Design. In The 49th Annual IEEE\/ACM International Symposium on Microarchitecture (MICRO-49)","author":"Rhu Minsoo","year":"2016","unstructured":"Minsoo Rhu, Natalia Gimelshein, Jason Clemons, Arslan Zulfiqar, and StephenW. Keckler. 2016. vDNN: Virtualized Deep Neural Networks for Scalable, Memoryefficient Neural Network Design. In The 49th Annual IEEE\/ACM International Symposium on Microarchitecture (MICRO-49)."},{"volume-title":"IEEE\/ACM International Symposium on Microarchitecture (MICRO).","author":"Rhu M.","key":"e_1_3_2_1_69_1","unstructured":"M. Rhu, N. Gimelshein, J. Clemons, A. Zulfiqar, and S. W. Keckler. 2016. vDNN: Virtualized deep neural networks for scalable, memory-efficient neural network design. In IEEE\/ACM International Symposium on Microarchitecture (MICRO)."},{"key":"e_1_3_2_1_70_1","volume-title":"Probabilistic Accuracy Bounds for Fault-tolerant Computations That Discard Tasks. In PInternational Conference on Supercomputing.","author":"Rinard Martin","year":"2006","unstructured":"Martin Rinard. 2006. Probabilistic Accuracy Bounds for Fault-tolerant Computations That Discard Tasks. In PInternational Conference on Supercomputing."},{"key":"e_1_3_2_1_71_1","volume-title":"Janghaeng Lee, and Scott Mahlke.","author":"Samadi Mehrzad","year":"2014","unstructured":"Mehrzad Samadi, Davoud Anoushe Jamshidi, Janghaeng Lee, and Scott Mahlke. 2014. Paraprox: Pattern-Based Approximation for Data Parallel Applications. In Architectural Support for Programming Languages and Operating Systems (ASPLOS)."},{"key":"e_1_3_2_1_72_1","doi-asserted-by":"publisher","DOI":"10.1145\/2540708.2540711"},{"key":"e_1_3_2_1_75_1","doi-asserted-by":"crossref","unstructured":"Adrian Sampson Werner Dietl Emily Fortuna Danushen Gnanapragasam Luis Ceze and Dan Grossman. 2011. EnerJ: Approximate Data Types for Safe and General Low-power Computation. In PLDI.","DOI":"10.1145\/1993498.1993518"},{"key":"e_1_3_2_1_76_1","doi-asserted-by":"crossref","unstructured":"Adrian Sampson Jacob Nelson Karin Strauss and Luis Ceze. 2014. Approximate Storage in Solid-state Memories. In ACM TOCS.","DOI":"10.1145\/2540708.2540712"},{"key":"e_1_3_2_1_77_1","doi-asserted-by":"publisher","DOI":"10.2514\/6.2007-5635"},{"key":"e_1_3_2_1_78_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.acha.2016.04.003"},{"key":"e_1_3_2_1_79_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.ijnonlinmec.2004.10.005"},{"key":"e_1_3_2_1_80_1","volume-title":"Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism. CoRR abs\/1909.08053","author":"Shoeybi Mohammad","year":"2019","unstructured":"Mohammad Shoeybi, Mostofa Patwary, Raul Puri, Patrick LeGresley, Jared Casper, and Bryan Catanzaro. 2019. Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism. CoRR abs\/1909.08053 (2019)."},{"key":"e_1_3_2_1_81_1","unstructured":"Mohammad Shoeybi Mostofa Patwary Raul Puri Patrick LeGresley Jared Casper and Bryan Catanzaro. 2019. Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism. arXiv:1909.08053 [cs.CL]"},{"key":"e_1_3_2_1_82_1","doi-asserted-by":"publisher","DOI":"10.1145\/2025113.2025133"},{"key":"e_1_3_2_1_83_1","doi-asserted-by":"publisher","DOI":"10.1080\/10556788"},{"key":"e_1_3_2_1_84_1","doi-asserted-by":"crossref","unstructured":"J. S. Smith O. Isayev and A. E. Roitberg. 2017. ANI-1: an extensible neural network potential with DFT accuracy at force field computational cost. (2017). Issue 4.","DOI":"10.1039\/C6SC05720A"},{"key":"e_1_3_2_1_85_1","doi-asserted-by":"publisher","DOI":"10.1145\/2954679.2872402"},{"key":"e_1_3_2_1_86_1","doi-asserted-by":"publisher","DOI":"10.5555\/3305890.3306035"},{"key":"e_1_3_2_1_87_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICTAI.2019.00209"},{"key":"e_1_3_2_1_88_1","doi-asserted-by":"publisher","DOI":"10.1145\/3178487.3178491"},{"key":"e_1_3_2_1_89_1","doi-asserted-by":"publisher","DOI":"10.1145\/3447786.3456251"},{"key":"e_1_3_2_1_90_1","doi-asserted-by":"publisher","DOI":"10.1145\/3447818.3460365"},{"key":"e_1_3_2_1_91_1","doi-asserted-by":"publisher","DOI":"10.1002\/cav.1695"},{"key":"e_1_3_2_1_92_1","volume-title":"Conference on Machine Learning and Systems.","author":"Yang Jie Amy","year":"2020","unstructured":"Jie Amy Yang, Jianyu Huang, Jongsoo Park, Ping Tak Peter Tang, and Andrew Tulloch. 2020. Mixed-Precision Embedding Using a Cache. In Conference on Machine Learning and Systems."},{"key":"e_1_3_2_1_93_1","doi-asserted-by":"publisher","DOI":"10.1109\/DLS49591.2019.00006"},{"key":"e_1_3_2_1_94_1","volume-title":"Betty: Enabling Large-Scale GNN Training with Batch-Level Graph Partitioning.","author":"Yang Shuangyan","year":"2023","unstructured":"Shuangyan Yang, Minjia Zhang, Wenqian Dong, and Dong Li. 2023. Betty: Enabling Large-Scale GNN Training with Batch-Level Graph Partitioning. (2023)."}],"event":{"name":"HPDC '23: The 32nd International Symposium on High-Performance Parallel and Distributed Computing","sponsor":["SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing","SIGARCH ACM Special Interest Group on Computer Architecture"],"location":"Orlando FL USA","acronym":"HPDC '23"},"container-title":["Proceedings of the 13th Workshop on AI and Scientific Computing at Scale using Flexible Computing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3589013.3596677","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3589013.3596677","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T16:36:15Z","timestamp":1750178175000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3589013.3596677"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,8,10]]},"references-count":92,"alternative-id":["10.1145\/3589013.3596677","10.1145\/3589013"],"URL":"https:\/\/doi.org\/10.1145\/3589013.3596677","relation":{},"subject":[],"published":{"date-parts":[[2023,8,10]]},"assertion":[{"value":"2023-08-11","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}