{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,9]],"date-time":"2026-08-09T13:17:38Z","timestamp":1786281458466,"version":"build-2736575974"},"reference-count":45,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2022,11,11]],"date-time":"2022-11-11T00:00:00Z","timestamp":1668124800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,11,11]],"date-time":"2022-11-11T00:00:00Z","timestamp":1668124800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100014188","name":"Ministry of Science and ICT, South Korea","doi-asserted-by":"publisher","award":["NRF-2021R1A2C1003379"],"award-info":[{"award-number":["NRF-2021R1A2C1003379"]}],"id":[{"id":"10.13039\/501100014188","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Cluster Comput"],"published-print":{"date-parts":[[2023,10]]},"DOI":"10.1007\/s10586-022-03805-x","type":"journal-article","created":{"date-parts":[[2022,11,11]],"date-time":"2022-11-11T18:03:54Z","timestamp":1668189834000},"page":"2835-2850","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Improving Oversubscribed GPU Memory Performance in the PyTorch Framework"],"prefix":"10.1007","volume":"26","author":[{"given":"Jake","family":"Choi","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Heon Young","family":"Yeom","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yoonhee","family":"Kim","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2022,11,11]]},"reference":[{"key":"3805_CR1","doi-asserted-by":"publisher","unstructured":"Ebubekir, B., Banu, D.: Performance Analysis and CPU vs GPU Comparison for Deep Learning. In: 2018 6th International Conference on Control Engineering & Information Technology (CEIT) (pp. 1\u20136). (2018). https:\/\/doi.org\/10.1109\/CEIT.2018.8751930","DOI":"10.1109\/CEIT.2018.8751930"},{"key":"3805_CR2","doi-asserted-by":"publisher","unstructured":"Huang, C., Jin, G., Li, J.: SwapAdvisor: Pushing Deep Learning Beyond the GPU Memory Limit via Smart Swapping. In Proceedings of the Twenty-Fifth International Conference on Architectural Support for Programming Languages and Operating Systems (ASPLOS \u201920). Association for Computing Machinery, New York, NY, USA, 1341\u20131355. (2020). https:\/\/doi.org\/10.1145\/3373376.3378530","DOI":"10.1145\/3373376.3378530"},{"key":"3805_CR3","unstructured":"Gupta, S., Agrawal, A., Gopalakrishnan, K., Narayanan, P.: Deep Learning with Limited Numerical Precision. In: Proceedings of the 32nd International Conference on International Conference on Machine Learning - Volume 37 (ICML\u201915). (2015). JMLR.org, 1737\u20131746"},{"key":"3805_CR4","doi-asserted-by":"crossref","unstructured":"Judd, P., Albericio, J., Hetherington, T., Aamodt, T., Jerger, N., Moshovos, A.: Proteus: Exploiting Numerical Precision Variability in Deep Neural Networks. In: Proceedings of the 2016 International Conference on Supercomputing (ICS\u201916). (2016). Association for Computing Machinery, Article 23","DOI":"10.1145\/2925426.2926294"},{"key":"3805_CR5","doi-asserted-by":"crossref","unstructured":"Chen, C., Choi, J., Brand, D., Agrawal, A., Zhang, W., Gopalakrishnan, K.: Adacomp: Adaptive residual gradient compression for data-parallel distributed training. In: AAAI (2018)","DOI":"10.1609\/aaai.v32i1.11728"},{"key":"3805_CR6","unstructured":"Lin, Y., Han, S., Mao, H., Wang, Y., Dally, W.: Deep Gradient Compression: Reducing the Communication Bandwidth for Distributed Training. arXiv preprint arXiv:1712.01887 (2017)"},{"key":"3805_CR7","unstructured":"Chen, T., Xu, B., Zhang, C., Guestrin, C.: Training deep nets with sublinear memory cost (2016). arXiv preprint arXiv:1604.06174 (2016)"},{"key":"3805_CR8","doi-asserted-by":"crossref","unstructured":"Rhu, M., Gimelshein, N., Clemons, J., Zulfiqar, A., Keckler, S.: vDNN: Virtualized Deep Neural Networks for Scalable, Memory-Efficient Neural Network Design. In: The 49th Annual IEEE\/ACM International Symposium on Microarchitecture (MICRO\u201949). IEEE Press, Article 18 (2016)","DOI":"10.1109\/MICRO.2016.7783721"},{"key":"3805_CR9","doi-asserted-by":"crossref","unstructured":"Jain, A., Phanishayee, A., Mars, J., Tang, L., Pekhimenko, G.: Gist: Efficient data encoding for deep neural network training. In 2018 ACM\/IEEE 45th Annual International Symposium on Computer Architecture (ISCA) (2018), IEEE, pp. 776\u2013789","DOI":"10.1109\/ISCA.2018.00070"},{"key":"3805_CR10","doi-asserted-by":"publisher","unstructured":"S. S.B., Garg, A., Kulkarni, P.: Dynamic Memory Management for GPU-Based Training of Deep Neural Networks. In: 2019 IEEE International Parallel and Distributed Processing Symposium (IPDPS), pp. 200\u2013209 (2019). https:\/\/doi.org\/10.1109\/IPDPS.2019.00030","DOI":"10.1109\/IPDPS.2019.00030"},{"key":"3805_CR11","doi-asserted-by":"crossref","unstructured":"Wang, L., Ye, J., Zhao, Y., Wu, W., Li, A., Leon Song, S., Xu, Z., Kraska, T.: Superneurons: Dynamic GPU Memory Management for Training Deep Neural Networks. In: Proceedings of the 23rd ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming (PPoPP\u201918) (2018)","DOI":"10.1145\/3178487.3178491"},{"key":"3805_CR12","unstructured":"Abadi, M., Barham, P., Chen, J., Chen, Z., A, Davis, Dean, J., Devin, M., Ghemawat, S., Irving, G., Isard, M., et al.: Tensorflow: A system for large-scale machine learning. In: OSDI, Vol. 16, pp. 265\u2013283 (2016)"},{"key":"3805_CR13","unstructured":"Collobert, R., Bengio, S., Mari\u00e9thoz, J.: Torch: a modular machine learning software library. Tech. rep, Idiap (2002)"},{"key":"3805_CR14","doi-asserted-by":"crossref","unstructured":"Jia, Y., Shelhamer, E., Donahue, J., Karayev, S., Long, J., Girshick, R., Guadarrama, S., Darrell, T.: Caffe: Convolutional architecture for fast feature embedding. In: Proceedings of the 22nd ACM international conference on Multimedia, ACM, pp. 675\u2013678 (2014)","DOI":"10.1145\/2647868.2654889"},{"key":"3805_CR15","unstructured":"Chen, T., Li, M., Li, Y., Lin, M., Wang, N., Wang, M., Xiao, T., Xu, B., Zhang, C., Zhang, Z.: Mxnet: A flexible and efficient machine learning library for heterogeneous distributed systems. arXiv preprint arXiv:1512.01274 (2015)"},{"key":"3805_CR16","unstructured":"Davis, L.: Handbook of genetic algorithms. (1991)"},{"key":"3805_CR17","doi-asserted-by":"crossref","unstructured":"Awan, A., Chu, C., Subramoni, H., Lu, X., Panda, D.: OCDNN: Exploiting Advanced Unified Memory Capabilities in CUDA 9 and Volta GPUs for Out-of-Core DNN Training. In: 25th IEEE International Conference on High Performance Computing, Data, and Analytics (HiPC) (2018)","DOI":"10.1109\/HiPC.2018.00024"},{"key":"3805_CR18","doi-asserted-by":"publisher","unstructured":"Manian, K.V., Ammar, A.A., Ruhela, A., Chu, C.-H., Subramoni, H., Panda, D. K.: Characterizing CUDA Unified Memory (UM)-Aware MPI Designs on Modern GPU Architectures. In: Proceedings of the 12th Workshop on General Purpose Processing Using GPUs (GPGPU \u201919). Association for Computing Machinery, New York, NY, USA, 43\u201352 (2019). https:\/\/doi.org\/10.1145\/3300053.3319419","DOI":"10.1145\/3300053.3319419"},{"key":"3805_CR19","unstructured":"Ren, J., Rajbhandari, S., R.Aminabadi, Y., Ruwase, O., Yang, S., Zhang, M., Li, D., He, Y.: ZeRO-Offload: Democratizing Billion-Scale Model Training. (2021). arXiv:abs\/2101.06840"},{"key":"3805_CR20","doi-asserted-by":"publisher","first-page":"7625","DOI":"10.1007\/s11227-019-02966-8","volume":"75","author":"M Knap","year":"2019","unstructured":"Knap, M., Czarnul, P.: Performance evaluation of Unified Memory with prefetching and oversubscription for selected parallel CUDA applications on NVIDIA Pascal and Volta GPUs. J. Supercomput. 75, 7625\u20137645 (2019). https:\/\/doi.org\/10.1007\/s11227-019-02966-8","journal-title":"J. Supercomput."},{"key":"3805_CR21","unstructured":"Sakharnykh, N.: Maximizing unified memory performance in cuda. (2017). https:\/\/devblogs.nvidia.com\/maximizing-unified-memory-performance-cuda\/"},{"key":"3805_CR22","doi-asserted-by":"publisher","unstructured":"Li, W., Jin, G., Cui, X., See, S.: An evaluation of unifed memory technology on nvidia gpus. In: 2015 15th IEEE\/ACM International Symposium on Cluster, Cloud and Grid Computing, pp 1092\u20131098. (2015). https:\/\/doi.org\/10.1109\/CCGrid.2015.105","DOI":"10.1109\/CCGrid.2015.105"},{"key":"3805_CR23","unstructured":"Paszke, A., Gross, S., Massa, F., Lerer, A., Bradbury, J., Chanan, G., Chintala, S.: PyTorch: An imperative style, high-performance deep learning library. In: Advances in Neural Information Processing Systems 32 (pp. 8024\u20138035). (2019). Curran Associates, Inc. Retrieved from http:\/\/papers.neurips.cc\/paper\/9015-pytorch-an-imperative-style-high-performance-deep-learning-library.pdf"},{"key":"3805_CR24","unstructured":"Theano Development Team. Theano: A Python framework for fast computation of mathematical expressions. arXiv e-prints, abs\/1605.02688, May (2016)"},{"key":"3805_CR25","doi-asserted-by":"publisher","unstructured":"Awan, A.A., Chu, C., Subramoni, H., Lu, X., Panda, D.K.: OC-DNN: Exploiting Advanced Unified Memory Capabilities in CUDA 9 and Volta GPUs for Out-of-Core DNN Training. In: 2018 IEEE 25th International Conference on High Performance Computing (HiPC), (2018), pp. 143\u2013152, https:\/\/doi.org\/10.1109\/HiPC.2018.00024","DOI":"10.1109\/HiPC.2018.00024"},{"key":"3805_CR26","unstructured":"Min, S., Wu, K., Huang, S., Hidayetoglu, M., Xiong, J., Ebrahimi, E., Chen, D., Hwu, W.: PyTorch-Direct: Enabling GPU Centric Data Access for Very Large Graph Neural Network Training with Irregular Accesses. CoRR abs\/2101.07956 (2021)"},{"key":"3805_CR27","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.E.: Imagenet classification with deep convolutional neural networks. In Advances in Neural Information Processing Systems; Curran Associates, Inc.: New York, NY, USA, (2012); pp. 1097\u20131105"},{"key":"3805_CR28","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, Las Vegas, NV, USA, 26 June\u20131 July (2016); pp. 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"3805_CR29","unstructured":"Simonyan, K., Zisserman, A.: Very deep convolutional networks for large-scale image recognition. arXiv 2014, arXiv:1409.1556"},{"key":"3805_CR30","unstructured":"Barnes, Z.: Techniques for Image Classification on Tiny-ImageNet. (2017)"},{"issue":"21","key":"3805_CR31","doi-asserted-by":"publisher","first-page":"10377","DOI":"10.3390\/app112110377","volume":"11","author":"H Choi","year":"2021","unstructured":"Choi, H., Lee, J.: Efficient use of GPU memory for large-scale deep learning model training. Appl. Sci. 11(21), 10377 (2021). https:\/\/doi.org\/10.3390\/app112110377","journal-title":"Appl. Sci."},{"key":"3805_CR32","doi-asserted-by":"crossref","unstructured":"Wolf, T., Debut, L., Sanh, V., Chaumond, J., Delangue, C., Moi, A., Cistac, P., Rault, T., Louf, R., Funtowicz, M., Davison, J., Shleifer, S., von Platen, P., Ma, C., Jernite, Y., Plu, J., Xu, C., Le Scao, T., Gugger, S., et al.: Transformers: state-of-the-art natural language processing. In: Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing: System Demonstrations, pp.\u00a038\u201345, Online. Association for Computational Linguistics. (2020)","DOI":"10.18653\/v1\/2020.emnlp-demos.6"},{"key":"3805_CR33","unstructured":"Merity, S., Xiong, C., Bradbury, J., Socher, R.: Pointer Sentinel Mixture Models. (2016)"},{"key":"3805_CR34","doi-asserted-by":"publisher","first-page":"1193","DOI":"10.1038\/s41467-021-21467-y","volume":"12","author":"CL Chen","year":"2021","unstructured":"Chen, C.L., Chen, C.C., Yu, W.H., et al.: An annotation-free whole-slide training approach to pathological classification of lung cancer types using deep learning. Nat. Commun. 12, 1193 (2021). https:\/\/doi.org\/10.1038\/s41467-021-21467-y","journal-title":"Nat. Commun."},{"key":"3805_CR35","doi-asserted-by":"publisher","DOI":"10.1038\/s41379-021-00838-2","author":"WY Chuang","year":"2021","unstructured":"Chuang, W.Y., Chen, C.C., Yu, W.H., et al.: Identification of nodal micrometastasis in colorectal cancer using deep learning on annotation-free whole-slide images. Mod. Pathol. (2021). https:\/\/doi.org\/10.1038\/s41379-021-00838-2","journal-title":"Mod. Pathol."},{"key":"3805_CR36","doi-asserted-by":"publisher","unstructured":"Choi, J., Yeom, H. Y., Kim, Y.: Implementing CUDA Unified Memory in the PyTorch Framework. In: 2021 IEEE International Conference on Autonomic Computing and Self-Organizing Systems Companion (ACSOS-C), (2021), pp. 20\u201325. https:\/\/doi.org\/10.1109\/ACSOS-C52956.2021.00029","DOI":"10.1109\/ACSOS-C52956.2021.00029"},{"key":"3805_CR37","unstructured":"Anaconda Software Distribution.: Anaconda Documentation. Anaconda Inc. Retrieved from https:\/\/docs.anaconda.com\/ (2020)"},{"key":"3805_CR38","unstructured":"Caffe2. https:\/\/caffe2.ai\/"},{"key":"3805_CR39","unstructured":"CUPTI. https:\/\/docs.nvidia.com\/cuda\/cupti\/index.html"},{"key":"3805_CR40","unstructured":"NVIDIA.: Beyond GPU Memory Limits with Unified Memory on Pascal, 2016. URL https:\/\/developer.nvidia.com\/blog\/beyond-gpumemory-limits-unified-memory-pascal\/"},{"key":"3805_CR41","unstructured":"NVIDIA, cuDNN: GPU Accelerated Deep Learning, 2016"},{"key":"3805_CR42","unstructured":"NVIDIA Profiler nvprof. https:\/\/docs.nvidia.com\/cuda\/profiler-users-guide\/index.html"},{"key":"3805_CR43","unstructured":"NVIDIA Profiler User\u2019s Guide. https:\/\/docs.nvidia.com\/cuda\/profiler-users-guide\/"},{"key":"3805_CR44","unstructured":"PyTorch Documentation.: https:\/\/pytorch.org\/docs\/ stable\/cpp_extension.html\/ (2020)"},{"key":"3805_CR45","unstructured":"CUDA-UVM-GPT2.: https:\/\/github.com\/kooyunmo\/cuda-uvm-gpt2\/ (2020)"}],"container-title":["Cluster Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10586-022-03805-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10586-022-03805-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10586-022-03805-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,8,26]],"date-time":"2023-08-26T20:24:28Z","timestamp":1693081468000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10586-022-03805-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,11,11]]},"references-count":45,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2023,10]]}},"alternative-id":["3805"],"URL":"https:\/\/doi.org\/10.1007\/s10586-022-03805-x","relation":{},"ISSN":["1386-7857","1573-7543"],"issn-type":[{"value":"1386-7857","type":"print"},{"value":"1573-7543","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,11,11]]},"assertion":[{"value":"28 April 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 October 2022","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 October 2022","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 November 2022","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no relevant financial or non-financial interests to disclose.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}},{"value":"Written informed consent for publication of this paper was obtained from all authors.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Informed Consent"}}]}}