{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,27]],"date-time":"2025-03-27T21:16:26Z","timestamp":1743110186359,"version":"3.40.3"},"publisher-location":"Singapore","reference-count":32,"publisher":"Springer Nature Singapore","isbn-type":[{"type":"print","value":"9789819981250"},{"type":"electronic","value":"9789819981267"}],"license":[{"start":{"date-parts":[[2023,11,13]],"date-time":"2023-11-13T00:00:00Z","timestamp":1699833600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,11,13]],"date-time":"2023-11-13T00:00:00Z","timestamp":1699833600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024]]},"DOI":"10.1007\/978-981-99-8126-7_30","type":"book-chapter","created":{"date-parts":[[2023,11,24]],"date-time":"2023-11-24T08:05:19Z","timestamp":1700813119000},"page":"383-395","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["A Memory Optimization Method for\u00a0Distributed Training"],"prefix":"10.1007","author":[{"given":"Tiantian","family":"Lv","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lu","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhigang","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chunxiao","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chuantao","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,11,13]]},"reference":[{"issue":"8","key":"30_CR1","first-page":"9","volume":"1","author":"A Radford","year":"2019","unstructured":"Radford, A., Wu, J., Child, R., Luan, D., Amodei, D., Sutskever, I., et al.: Language models are unsupervised multitask learners. OpenAI Blog 1(8), 9 (2019)","journal-title":"OpenAI Blog"},{"key":"30_CR2","unstructured":"Vaswani, A., et al.: Attention is all you need. In: Advances in Neural Information Processing Systems, vol. 30 (2017)"},{"key":"30_CR3","unstructured":"Cai, H., Zhu, L., Han, S.: Proxylessnas: direct neural architecture search on target task and hardware. arXiv preprint arXiv:1812.00332 (2018)"},{"key":"30_CR4","unstructured":"Huang, Y., et al.: GPipe: efficient training of giant neural networks using pipeline parallelism. In: Advances in Neural Information Processing Systems, vol. 32 (2019)"},{"key":"30_CR5","unstructured":"Alvarez, J.M., Salzmann, M.: Learning the number of neurons in deep networks. In: Advances in Neural Information Processing Systems, vol. 29 (2016)"},{"key":"30_CR6","unstructured":"Han, S., Pool, J., Tran, J., Dally, W.: Learning both weights and connections for efficient neural network. In: Advances in Neural Information Processing Systems, vol. 28 (2015)"},{"issue":"6","key":"30_CR7","doi-asserted-by":"publisher","first-page":"84","DOI":"10.1145\/3065386","volume":"60","author":"A Krizhevsky","year":"2017","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.E.: Imagenet classification with deep convolutional neural networks. Commun. ACM 60(6), 84\u201390 (2017)","journal-title":"Commun. ACM"},{"key":"30_CR8","unstructured":"Goyal, P., et al.: Accurate, large minibatch SGD: training imagenet in 1 hour. arXiv preprint arXiv:1706.02677 (2017)"},{"key":"30_CR9","unstructured":"Shallue, C.J., Lee, J., Antognini, J., Sohl-Dickstein, J., Frostig R., Dahl, G.E.: Measuring the effects of data parallelism on neural network training. arXiv preprint arXiv:1811.03600 (2018)"},{"key":"30_CR10","unstructured":"Sabet, M.J., Dufter, P., Yvon, F., Sch\u00fctze, H.: Simalign: high quality word alignments without parallel training data using static and contextualized embeddings. arXiv preprint arXiv:2004.08728 (2020)"},{"key":"30_CR11","doi-asserted-by":"crossref","unstructured":"Fan, S., et al.: Dapple: a pipelined data parallel approach for training large models. In: Proceedings of the 26th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming, pp. 431\u2013445 (2021)","DOI":"10.1145\/3437801.3441593"},{"key":"30_CR12","doi-asserted-by":"publisher","first-page":"1290","DOI":"10.1109\/TASLP.2021.3066047","volume":"29","author":"M Zhang","year":"2021","unstructured":"Zhang, M., Zhou, Y., Zhao, L., Li, H.: Transfer learning from speech synthesis to voice conversion with non-parallel training data. IEEE\/ACM Trans. Audio Speech Lang. Process. 29, 1290\u20131302 (2021)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"30_CR13","unstructured":"Dean, J., et al.: Large scale distributed deep networks. In: Advances in Neural Information Processing Systems, vol. 25 (2012)"},{"key":"30_CR14","unstructured":"Shoeybi, M., Patwary, M., Puri, R., LeGresley, P., Casper, J., Catanzaro, B.: Megatron-LM: training multi-billion parameter language models using model parallelism. arXiv preprint arXiv:1909.08053 (2019)"},{"key":"30_CR15","first-page":"1","volume":"1","author":"Z Jia","year":"2019","unstructured":"Jia, Z., Zaharia, M., Aiken, A.: Beyond data and model parallelism for deep neural networks. Proc. Mach. Learn. Syst. 1, 1\u201313 (2019)","journal-title":"Proc. Mach. Learn. Syst."},{"key":"30_CR16","unstructured":"Narayanan, D., Phanishayee, A., Shi, K., Chen, X., Zaharia, M.: Memory-efficient pipeline-parallel DNN training. In: International Conference on Machine Learning, pp. 7937\u20137947. PMLR (2021)"},{"key":"30_CR17","unstructured":"Moritz, P., et al.: Ray: a distributed framework for emerging $$\\{$$AI$$\\}$$ applications. In: 13th $$\\{$$USENIX$$\\}$$ Symposium on Operating Systems Design and Implementation ($$\\{$$OSDI$$\\}$$ 18), pp. 561\u2013577 (2018)"},{"key":"30_CR18","doi-asserted-by":"crossref","unstructured":"Narayanan, D., et al.: Pipedream: generalized pipeline parallelism for DNN training. In: Proceedings of the 27th ACM Symposium on Operating Systems Principles, pp. 1\u201315 (2019)","DOI":"10.1145\/3341301.3359646"},{"key":"30_CR19","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/s42452-021-04897-7","volume":"4","author":"Q Zhang","year":"2022","unstructured":"Zhang, Q.: A novel resnet101 model based on dense dilated convolution for image classification. SN Appl. Sci. 4, 1\u201313 (2022)","journal-title":"SN Appl. Sci."},{"key":"30_CR20","doi-asserted-by":"crossref","unstructured":"Real, E., Aggarwal, A., Huang, Y., Le, Q.V.: Regularized evolution for image classifier architecture search. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 33, no. 01, pp. 4780\u20134789 (2019)","DOI":"10.1609\/aaai.v33i01.33014780"},{"key":"30_CR21","doi-asserted-by":"crossref","unstructured":"Jiang, J., Cui, B., Zhang, C., Yu, L.: Heterogeneity-aware distributed parameter servers. In: Proceedings of the 2017 ACM International Conference on Management of Data, pp. 463\u2013478 (2017)","DOI":"10.1145\/3035918.3035933"},{"key":"30_CR22","unstructured":"Jia, Z., Lin, S., Qi, C.R., Aiken, A.: Exploring hidden dimensions in accelerating convolutional neural networks. In: International Conference on Machine Learning, pp. 2274\u20132283. PMLR (2018)"},{"key":"30_CR23","doi-asserted-by":"crossref","unstructured":"Jiang, W., et al.: A novel stochastic gradient descent algorithm based on grouping over heterogeneous cluster systems for distributed deep learning. In: 2019 19th IEEE\/ACM International Symposium on Cluster, Cloud and Grid Computing (CCGRID), pp. 391\u2013398. IEEE (2019)","DOI":"10.1109\/CCGRID.2019.00053"},{"key":"30_CR24","doi-asserted-by":"crossref","unstructured":"Kim, J.K., et al.: STRADS: a distributed framework for scheduled model parallel machine learning. In: Proceedings of the Eleventh European Conference on Computer Systems, pp. 1\u201316 (2016)","DOI":"10.1145\/2901318.2901331"},{"issue":"8","key":"30_CR25","doi-asserted-by":"publisher","first-page":"772","DOI":"10.1038\/nmeth.2109","volume":"9","author":"D Darriba","year":"2012","unstructured":"Darriba, D., Taboada, G.L., Doallo, R., Posada, D.: jmodeltest 2: more models, new heuristics and parallel computing. Nat. Methods 9(8), 772\u2013772 (2012)","journal-title":"Nat. Methods"},{"issue":"5","key":"30_CR26","first-page":"1017","volume":"71","author":"L Prosky","year":"1988","unstructured":"Prosky, L., Asp, N.-G., Schweizer, T.F., Devries, J.W., Furda, I.: Determination of insoluble, soluble, and total dietary fiber in foods and food products: interlaboratory study. J. Assoc. Off. Anal. Chem. 71(5), 1017\u20131023 (1988)","journal-title":"J. Assoc. Off. Anal. Chem."},{"key":"30_CR27","doi-asserted-by":"crossref","unstructured":"L. Shen, Y. Mao, Z. Wang, H. Nie, and J. Huang, \"Dnn training optimization with pipelined parallel based on feature maps encoding. In: 2022 Tenth International Conference on Advanced Cloud and Big Data (CBD), pp. 36\u201341. IEEE (2022)","DOI":"10.1109\/CBD58033.2022.00016"},{"key":"30_CR28","unstructured":"Chen, C.-C., Yang, C.-L., Cheng,H.-Y.: Efficient and robust parallel DNN training through model parallelism on multi-GPU platform. arXiv preprint arXiv:1809.02839 (2018)"},{"key":"30_CR29","unstructured":"Yang, B., Zhang, J., Li, J., R\u00e9, C., Aberger, C., De Sa, C.: PipeMare: asynchronous pipeline parallel DNN training. In: Proceedings of Machine Learning and Systems, vol. 3, pp. 269\u2013296 (2021)"},{"issue":"39","key":"30_CR30","first-page":"851","volume":"3","author":"M Harris","year":"2007","unstructured":"Harris, M., Sengupta, S., Owens, J.D.: Parallel prefix sum (scan) with cuda. GPU Gems 3(39), 851\u2013876 (2007)","journal-title":"GPU Gems"},{"key":"30_CR31","unstructured":"Sengupta, S., Lefohn, A., Owens, J.D.: \" work-efficient step-efficient prefix sum algorithm (2006)"},{"key":"30_CR32","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"170","DOI":"10.1007\/978-3-030-55754-6_10","volume-title":"NASA Formal Methods","author":"M Safari","year":"2020","unstructured":"Safari, M., Oortwijn, W., Joosten, S., Huisman, M.: Formal verification of parallel prefix sum. In: Lee, R., Jha, S., Mavridou, A., Giannakopoulou, D. (eds.) NFM 2020. LNCS, vol. 12229, pp. 170\u2013186. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-55754-6_10"}],"container-title":["Communications in Computer and Information Science","Neural Information Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-99-8126-7_30","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,7]],"date-time":"2024-03-07T11:36:15Z","timestamp":1709811375000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-99-8126-7_30"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,11,13]]},"ISBN":["9789819981250","9789819981267"],"references-count":32,"URL":"https:\/\/doi.org\/10.1007\/978-981-99-8126-7_30","relation":{},"ISSN":["1865-0929","1865-0937"],"issn-type":[{"type":"print","value":"1865-0929"},{"type":"electronic","value":"1865-0937"}],"subject":[],"published":{"date-parts":[[2023,11,13]]},"assertion":[{"value":"13 November 2023","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICONIP","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Neural Information Processing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Changsha","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2023","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"20 November 2023","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 November 2023","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"30","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"iconip2023","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/iconip2023.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Single-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"EasyChair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"1274","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"650","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"51% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"4.14","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"2.46","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}