{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,26]],"date-time":"2025-03-26T05:45:25Z","timestamp":1742967925796,"version":"3.40.3"},"publisher-location":"Cham","reference-count":35,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783030633929"},{"type":"electronic","value":"9783030633936"}],"license":[{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2020]]},"DOI":"10.1007\/978-3-030-63393-6_8","type":"book-chapter","created":{"date-parts":[[2020,12,22]],"date-time":"2020-12-22T09:04:28Z","timestamp":1608627868000},"page":"117-129","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":7,"title":["DataStates: Towards Lightweight Data Models for Deep Learning"],"prefix":"10.1007","author":[{"given":"Bogdan","family":"Nicolae","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2020,12,18]]},"reference":[{"key":"8_CR1","unstructured":"Abadi, M., et al.: TensorFlow: large-scale machine learning on heterogeneous systems (2015). http:\/\/tensorflow.org\/"},{"issue":"160018","key":"8_CR2","first-page":"1","volume":"3","author":"MD Wilkinson","year":"2016","unstructured":"Wilkinson, M.D., et al.: The fair guiding principles for scientific data management and stewardship. Sci. Data 3(160018), 1\u20139 (2016)","journal-title":"Sci. Data"},{"key":"8_CR3","doi-asserted-by":"crossref","unstructured":"Balaprakash, P., et al.: Scalable reinforcement-learning-based neural architecture search for cancer deep learning research. In: The 2019 International Conference for High Performance Computing, Networking, Storage and Analysis, SC 2019, pp. 37:1\u201337:33 (2019)","DOI":"10.1145\/3295500.3356202"},{"key":"8_CR4","doi-asserted-by":"crossref","unstructured":"Bautista-Gomez, L., Tsuboi, S., Komatitsch, D., Cappello, F., Maruyama, N., Matsuoka, S.: FTI: High performance fault tolerance interface for hybrid systems. In: The 2011 ACM\/IEEE International Conference for High Performance Computing, Networking, Storage and Analysis, SC 2011, Seattle, USA, pp. 32:1\u201332:32 (2011)","DOI":"10.1145\/2063384.2063427"},{"key":"8_CR5","first-page":"212","volume":"2011","author":"J Bernard","year":"2011","unstructured":"Bernard, J.: Mercurial-revision control approximated. Linux J. 2011, 212 (2011)","journal-title":"Linux J."},{"key":"8_CR6","volume-title":"High Performance In-memory Computing with Apache Ignite","author":"S Bhuiyan","year":"2017","unstructured":"Bhuiyan, S., Zheludkov, M., Isachenko, T.: High Performance In-memory Computing with Apache Ignite. Lulu Press, Morrisville (2017). https:\/\/www.lulu.com\/"},{"key":"8_CR7","unstructured":"Cao, L., Settlemyer, B.W., Bent, J.: To share or not to share: comparing burst buffer architectures. In: The 25th High Performance Computing Symposium, HPC 2017, Virginia Beach, Virginia, pp. 4:1\u20134:10 (2017)"},{"key":"8_CR8","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4842-0076-6","volume-title":"Pro Git","author":"S Chacon","year":"2014","unstructured":"Chacon, S., Straub, B.: Pro Git, 2nd edn. Apress, Berkely (2014)","edition":"2"},{"key":"8_CR9","doi-asserted-by":"crossref","unstructured":"Chard, R., et al.: Publishing and serving machine learning models with DLHub. In: Practice and Experience in Advanced Research Computing on Rise of the Machines (Learning), PEARC 2019, Chicago, USA (2019)","DOI":"10.1145\/3332186.3332246"},{"key":"8_CR10","unstructured":"Collins-Sussman, B.: The subversion project: buiding a better CVS. Linux J. 2002(94) (2002)"},{"key":"8_CR11","unstructured":"Dean, J., et al.: Large scale distributed deep networks. In: The 25th International Conference on Neural Information Processing Systems, NIPS 2012, Lake Tahoe, USA, pp. 1223\u20131231 (2012)"},{"key":"8_CR12","doi-asserted-by":"crossref","unstructured":"Lawson, M., et al.: Empress: extensible metadata provider for extreme-scale scientific simulations. In: The 2nd Joint International Workshop on Parallel Data Storage and Data Intensive Scalable Computing Systems, PDSW-DISCS@SC 2017, pp. 19\u201324 (2017)","DOI":"10.1145\/3149393.3149403"},{"key":"8_CR13","doi-asserted-by":"crossref","unstructured":"Li, J., Nicolae, B., Wozniak, J., Bosilca, G.: Understanding scalability and fine-grain parallelism of synchronous data parallel training. In: 5th Workshop on Machine Learning in HPC Environments (in Conjunction with SC19), MLHPC 2019, Denver, USA, pp. 1\u20138 (2019)","DOI":"10.1109\/MLHPC49564.2019.00006"},{"key":"8_CR14","unstructured":"Lockwood, G., et al.: Storage 2020: a vision for the future of HPC storage. Technical Report, Lawrence Berkeley National Laboratory (2017)"},{"key":"8_CR15","doi-asserted-by":"crossref","unstructured":"Lofstead, J., Baker, J., Younge, A.: Data pallets: containerizing storage for reproducibility and traceability. In: 2019 International Conference on High Performance Computing, ISC 2019, pp. 36\u201345 (2019)","DOI":"10.1007\/978-3-030-34356-9_4"},{"key":"8_CR16","doi-asserted-by":"crossref","unstructured":"Lofstead, J., Jimenez, I., Maltzahn, C., Koziol, Q., Bent, J., Barton, E.: DAOS and friends: a proposal for an exascale storage system. In: The 2016 International Conference for High Performance Computing, Networking, Storage and Analysis, SC 2016, Salt Lake City, Utah, pp. 50:1\u201350:12 (2016)","DOI":"10.1109\/SC.2016.49"},{"issue":"239","key":"8_CR17","first-page":"2","volume":"2014","author":"D Merkel","year":"2014","unstructured":"Merkel, D.: Docker: lightweight Linux containers for consistent development and deployment. Linux J. 2014(239), 2 (2014)","journal-title":"Linux J."},{"key":"8_CR18","doi-asserted-by":"crossref","unstructured":"Moody, A., Bronevetsky, G., Mohror, K., Supinski, B.R.D.: Design, modeling, and evaluation of a scalable multi-level checkpointing system. In: The 2010 ACM\/IEEE International Conference for High Performance Computing, Networking, Storage and Analysis, SC 2010, New Orleans, USA, pp. 1:1\u20131:11 (2010)","DOI":"10.1109\/SC.2010.18"},{"key":"8_CR19","doi-asserted-by":"crossref","unstructured":"Narayanan, D., et al.: PipeDream: generalized pipeline parallelism for DNN training. In: The 27th ACM Symposium on Operating Systems Principles, SOSP 2019, Huntsville, Canada, pp. 1\u201315 (2019)","DOI":"10.1145\/3341301.3359646"},{"key":"8_CR20","unstructured":"Nicolae, B.: Towards scalable checkpoint restart: a collective inline memory contents deduplication proposal. In: The 27th IEEE International Parallel and Distributed Processing Symposium, IPDPS 2013, Boston, USA (2013). http:\/\/hal.inria.fr\/hal-00781532\/en"},{"key":"8_CR21","doi-asserted-by":"crossref","unstructured":"Nicolae, B.: Leveraging naturally distributed data redundancy to reduce collective I\/O replication overhead. In: 29th IEEE International Parallel and Distributed Processing Symposium, IPDPS 2015, pp. 1023\u20131032 (2015)","DOI":"10.1109\/IPDPS.2015.82"},{"key":"8_CR22","doi-asserted-by":"publisher","first-page":"169","DOI":"10.1016\/j.jpdc.2010.08.004","volume":"71","author":"B Nicolae","year":"2011","unstructured":"Nicolae, B., Antoniu, G., Boug\u00e9, L., Moise, D., Carpen-Amarie, A.: BlobSeer: next-generation data management for large scale infrastructures. J. Parallel Distrib. Comput. 71, 169\u2013184 (2011)","journal-title":"J. Parallel Distrib. Comput."},{"key":"8_CR23","doi-asserted-by":"crossref","unstructured":"Nicolae, B., Li, J., Wozniak, J., Bosilca, G., Dorier, M., Cappello, F.: DeepFreeze: towards scalable asynchronous checkpointing of deep learning models. In: 20th IEEE\/ACM International Symposium on Cluster, Cloud and Internet Computing, CGrid 2020, Melbourne, Australia, pp. 172\u2013181 (2020)","DOI":"10.1109\/CCGrid49817.2020.00-76"},{"key":"8_CR24","doi-asserted-by":"crossref","unstructured":"Nicolae, B., Moody, A., Gonsiorowski, E., Mohror, K., Cappello, F.: VeloC: towards high performance adaptive asynchronous checkpointing at large scale. In: The 2019 IEEE International Parallel and Distributed Processing Symposium, IPDPS 2019, Rio de Janeiro, Brazil, pp. 911\u2013920 (2019)","DOI":"10.1109\/IPDPS.2019.00099"},{"key":"8_CR25","doi-asserted-by":"crossref","unstructured":"Nicolae, B., Wozniak, J.M., Dorier, M., Cappello, F.: DeepClone: lightweight state replication of deep learning models for data parallel training. In: The 2020 IEEE International Conference on Cluster Computing, CLUSTER 2020, Kobe, Japan (2020)","DOI":"10.1109\/CLUSTER49012.2020.00033"},{"key":"8_CR26","unstructured":"Real, E., et al.: Large-scale evolution of image classifiers. In: The 34th International Conference on Machine Learning, ICML 2017, Sydney, Australia, pp. 2902\u20132911 (2017)"},{"key":"8_CR27","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"260","DOI":"10.1007\/978-3-319-58943-5_21","volume-title":"Euro-Par 2016: Parallel Processing Workshops","author":"N Saurabh","year":"2017","unstructured":"Saurabh, N., Kimovski, D., Ostermann, S., Prodan, R.: VM image repository and distribution models for federated clouds: state of the art, possible directions and open issues. In: Desprez, F., et al. (eds.) Euro-Par 2016. LNCS, vol. 10104, pp. 260\u2013271. Springer, Cham (2017). https:\/\/doi.org\/10.1007\/978-3-319-58943-5_21"},{"key":"8_CR28","doi-asserted-by":"crossref","unstructured":"Shanahan, J.G., Dai, L.: Large scale distributed data science using apache spark. In: The 21th ACM SIGKDD International Conference on Knowledge Discovery and Data Mining, KDD 2015, Sydney, Australia, pp. 2323\u20132324 (2015)","DOI":"10.1145\/2783258.2789993"},{"key":"8_CR29","doi-asserted-by":"crossref","unstructured":"Shu, H., Zhu, H.: Sensitivity analysis of deep neural networks. In: The 33rd AAAI Conference of Artificial Intelligence, AAAI 2019, pp. 4943\u20134950 (2019)","DOI":"10.1609\/aaai.v33i01.33014943"},{"key":"8_CR30","doi-asserted-by":"crossref","unstructured":"Teerapittayanon, S., McDanel, B., Kung, H.T.: BranchyNet: fast inference via early exiting from deep neural networks. In: The 23rd International Conference on Pattern Recognition, ICPR 2016, Cancun, Mexico, pp. 2464\u20132469 (2016)","DOI":"10.1109\/ICPR.2016.7900006"},{"key":"8_CR31","doi-asserted-by":"crossref","unstructured":"Tseng, S.M., Nicolae, B., Bosilca, G., Jeannot, E., Cappello, F.: Towards portable online prediction of network utilization using MPI-level monitoring. In: 25th International European Conference on Parallel and Distributed Systems, EuroPar 2019, Goettingen, Germany, pp. 1\u201314 (2019)","DOI":"10.1007\/978-3-030-29400-7_4"},{"issue":"491","key":"8_CR32","first-page":"59","volume":"19","author":"J Wozniak","year":"2018","unstructured":"Wozniak, J., et al.: CANDLE\/supervisor: A workflow framework for machine learning applied to cancer research. BMC Bioinform. 19(491), 59\u201369 (2018)","journal-title":"BMC Bioinform."},{"key":"8_CR33","unstructured":"Zaharia, M., et al.: Resilient distributed datasets: a fault-tolerant abstraction for in-memory cluster computing. In: Proceedings of the 9th USENIX Conference on Networked Systems Design and Implementation, NSDI 2012, San Jose, USA, p. 2 (2012)"},{"key":"8_CR34","unstructured":"Zaharia, M., Chowdhury, M., Franklin, M.J., Shenker, S., Stoica, I.: Spark: cluster computing with working sets. In: The 2Nd USENIX Conference on Hot Topics in Cloud Computing, HotCloud 2010, Boston, MA, p. 10 (2010)"},{"key":"8_CR35","unstructured":"Zhang, S., Boehmer, W., Whiteson, S.: Deep residual reinforcement learning. In: The 19th International Conference on Autonomous Agents and MultiAgent Systems, AAMAS 2020, Auckland, New Zealand, pp. 1611\u20131619 (2020)"}],"container-title":["Communications in Computer and Information Science","Driving Scientific and Engineering Discoveries Through the Convergence of HPC, Big Data and AI"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-63393-6_8","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,12,8]],"date-time":"2022-12-08T12:28:17Z","timestamp":1670502497000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-030-63393-6_8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020]]},"ISBN":["9783030633929","9783030633936"],"references-count":35,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-63393-6_8","relation":{},"ISSN":["1865-0929","1865-0937"],"issn-type":[{"type":"print","value":"1865-0929"},{"type":"electronic","value":"1865-0937"}],"subject":[],"published":{"date-parts":[[2020]]},"assertion":[{"value":"18 December 2020","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"SMC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Smoky Mountains Computational Sciences and Engineering Conference","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Oak Ridge, TN","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"USA","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2020","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 August 2020","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28 August 2020","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"smc2020","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/smc.ornl.gov\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Single-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"easychair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"94","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"36","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"1","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"38% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.75","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"No","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"The conference was held virtually due to the COVID-19 pandemic.","order":10,"name":"additional_info_on_review_process","label":"Additional Info on Review Process","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}