{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T20:21:30Z","timestamp":1740169290014,"version":"3.37.3"},"reference-count":62,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"DOI":"10.13039\/501100001321","name":"NRF","doi-asserted-by":"publisher","award":["NRF-2019K1A3A1A77074958"],"award-info":[{"award-number":["NRF-2019K1A3A1A77074958"]}],"id":[{"id":"10.13039\/501100001321","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100014188","name":"Korea Government [Ministry of Science and ICT (MSIT)]","doi-asserted-by":"publisher","award":["IITP-2021-0-01341"],"award-info":[{"award-number":["IITP-2021-0-01341"]}],"id":[{"id":"10.13039\/501100014188","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Artificial Intelligence Graduate School, Chung-Ang University"},{"DOI":"10.13039\/501100014188","name":"High-Potential Individuals Global Training Program","doi-asserted-by":"publisher","award":["IITP-2021-0-01574"],"award-info":[{"award-number":["IITP-2021-0-01574"]}],"id":[{"id":"10.13039\/501100014188","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Access"],"published-print":{"date-parts":[[2022]]},"DOI":"10.1109\/access.2022.3171910","type":"journal-article","created":{"date-parts":[[2022,5,2]],"date-time":"2022-05-02T20:22:02Z","timestamp":1651522922000},"page":"34706-34715","source":"Crossref","is-referenced-by-count":2,"title":["Regularization in Network Optimization via Trimmed Stochastic Gradient Descent With Noisy Label"],"prefix":"10.1109","volume":"10","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6858-3551","authenticated-orcid":false,"given":"Kensuke","family":"Nakamura","sequence":"first","affiliation":[{"name":"Computer Science Department, Chung-Ang University, Seoul, South Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bong-Soo","family":"Sohn","sequence":"additional","affiliation":[{"name":"Computer Science Department, Chung-Ang University, Seoul, South Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kyoung-Jae","family":"Won","sequence":"additional","affiliation":[{"name":"Biotech Research and Innovation Centre (BRIC), University of Copenhagen, Copenhagen, Denmark"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2752-3939","authenticated-orcid":false,"given":"Byung-Woo","family":"Hong","sequence":"additional","affiliation":[{"name":"Computer Science Department, Chung-Ang University, Seoul, South Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1201\/9781420011814"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1145\/3446776"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.308"},{"key":"ref4","article-title":"Very deep convolutional networks for large-scale image recognition","author":"Simonyan","year":"2014","journal-title":"arXiv:1409.1556"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.195"},{"key":"ref6","first-page":"2171","article-title":"Scalable Bayesian optimization using deep neural networks","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Snoek"},{"key":"ref7","first-page":"1168","article-title":"Unbounded Bayesian optimization via regularization","volume-title":"Proc. Artif. Intell. Statist.","author":"Shahriari"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-75786-5_23"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-35289-8_26"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2019.2937139"},{"issue":"1","key":"ref11","first-page":"1929","article-title":"Dropout: A simple way to prevent neural networks from overfitting","volume":"15","author":"Srivastava","year":"2014","journal-title":"J. Mach. Learn. Res."},{"key":"ref12","first-page":"3084","article-title":"Adaptive dropout for training deep neural networks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Ba"},{"key":"ref13","first-page":"2575","article-title":"Variational dropout and the local reparameterization trick","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Kingma"},{"key":"ref14","article-title":"Generalized dropout","author":"Srinivas","year":"2016","journal-title":"arXiv:1611.06791"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46493-0_39"},{"key":"ref16","first-page":"5109","article-title":"Regularizing deep neural networks by noise: Its interpretation and optimization","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Noh"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1214\/aoms\/1177729586"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.7551\/mitpress\/1888.003.0013"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1145\/1015330.1015332"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-7908-2604-3_16"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1137\/16M1080173"},{"key":"ref22","first-page":"7654","article-title":"The anisotropic noise in stochastic gradient descent: Its behavior of escaping from sharp minima and regularization effects","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Zhu"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1007\/s10994-006-8365-9"},{"key":"ref24","first-page":"1","article-title":"Don\u2019t decay the learning rate, increase the batch size","volume-title":"Proc. Int. Conf. Learn. Represent. (ICLR)","author":"Smith"},{"key":"ref25","first-page":"2121","article-title":"Adaptive subgradient methods for online learning and stochastic optimization","volume":"12","author":"Duchi","year":"2011","journal-title":"J. Mach. Learn. Res."},{"issue":"2","key":"ref26","first-page":"26","article-title":"Lecture 6.5-RMSPROP: Divide the gradient by a running average of its recent magnitude","volume":"4","author":"Tieleman","year":"2012","journal-title":"COURSERA, Neural Netw. Mach. Learn."},{"key":"ref27","article-title":"Adam: A method for stochastic optimization","author":"Kingma","year":"2014","journal-title":"arXiv:1412.6980"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1007\/3-540-49430-8_3"},{"key":"ref29","first-page":"1223","article-title":"Large scale distributed deep networks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Dean"},{"key":"ref30","article-title":"Accurate, large minibatch SGD: Training ImageNet in 1 hour","author":"Goyal","year":"2017","journal-title":"arXiv:1706.02677"},{"key":"ref31","first-page":"4613","article-title":"Byzantine stochastic gradient descent","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Alistarh"},{"key":"ref32","article-title":"Byzantine-robust distributed learning: Towards optimal statistical rates","author":"Yin","year":"2018","journal-title":"arXiv:1803.01498"},{"key":"ref33","article-title":"Distributed training with heterogeneous data: Bridging median- and mean-based algorithms","author":"Chen","year":"2019","journal-title":"arXiv:1906.01736"},{"key":"ref34","article-title":"Outlier robust online learning","author":"Feng","year":"2017","journal-title":"arXiv:1701.00251"},{"key":"ref35","article-title":"Learning with bad training data via iterative trimmed loss minimization","author":"Shen","year":"2018","journal-title":"arXiv:1810.11874"},{"key":"ref36","first-page":"1984","article-title":"Efficient stochastic gradient hard thresholding","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Zhou"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-73074-5_8"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.1.1"},{"key":"ref39","first-page":"1","article-title":"Entropy-SGD: Biasing gradient descent into wide valleys","volume-title":"Proc. Int. Conf. Learn. Represent. (ICLR)","author":"Chaudhari"},{"key":"ref40","first-page":"1019","article-title":"Sharp minima can generalize for deep nets","volume-title":"Proc. 34th Int. Conf. Mach. Learn.","volume":"70","author":"Dinh"},{"key":"ref41","first-page":"2663","article-title":"A stochastic gradient method with an exponential convergence rate for finite training sets","volume-title":"Proc. 25th Int. Conf. Neural Inf. Process. Syst. (NIPS)","volume":"2","author":"Le Roux"},{"key":"ref42","first-page":"315","article-title":"Accelerating stochastic gradient descent using predictive variance reduction","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Johnson"},{"key":"ref43","first-page":"764","article-title":"On the theory of variance reduction for stochastic gradient Monte Carlo","volume-title":"Proc. ICML","author":"Chatterji"},{"key":"ref44","first-page":"46","article-title":"Fast stochastic alternating direction method of multipliers","volume-title":"Proc. 31st Int. Conf. Mach. Learn. (ICML)","author":"Zhong"},{"key":"ref45","first-page":"1990","article-title":"Adaptive variance reducing for stochastic gradient descent","volume-title":"Proc. IJCAI","author":"Shen"},{"key":"ref46","first-page":"6028","article-title":"Stochastic variance-reduced Hamilton Monte Carlo methods","volume-title":"Proc. 35th Int. Conf. Mach. Learn.","volume":"80","author":"Zou"},{"key":"ref47","first-page":"5980","article-title":"A simple stochastic variance reduced algorithm with fast convergence rates","volume-title":"Proc. 35th Int. Conf. Mach. Learn.","volume":"80","author":"Zhou"},{"key":"ref48","first-page":"699","article-title":"Variance reduction for faster non-convex optimization","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Allen-Zhu"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v31i1.10940"},{"key":"ref50","first-page":"3731","article-title":"Zeroth-order stochastic variance reduction for nonconvex optimization","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Liu"},{"key":"ref51","first-page":"823","article-title":"Two problems with back propagation and other steepest descent learning procedures for networks","volume-title":"Proc. 8th Annu. Conf. Cognit. Sci. Soc.","author":"Sutton"},{"key":"ref52","first-page":"543","article-title":"A method for unconstrained convex minimization problem with the rate of convergence $O(1\/k^{2})$","volume-title":"Proc. Doklady AN USSR","volume":"269","author":"Nesterov"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/ITA.2018.8503173"},{"key":"ref54","first-page":"797","article-title":"Escaping from saddle points\u2014Online stochastic gradient for tensor decomposition","volume-title":"Proc. Conf. Learn. Theory","author":"Ge"},{"key":"ref55","first-page":"1724","article-title":"How to escape saddle points efficiently","volume-title":"Proc. 34th Int. Conf. Mach. Learn.","volume":"70","author":"Jin"},{"key":"ref56","first-page":"2698","article-title":"An alternative view: When does SGD escape local minima?","volume-title":"Proc. ICML","author":"Kleinberg"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1016\/0893-6080(91)90047-9"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/5.726791"},{"key":"ref59","article-title":"Fashion-MNIST: A novel image dataset for benchmarking machine learning algorithms","author":"Xiao","year":"2017","journal-title":"arXiv:1708.07747"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN.2017.7966217"},{"key":"ref61","first-page":"1","article-title":"Time matters in regularizing deep networks: Weight decay and data augmentation affect early learning dynamics, matter little near convergence","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"32","author":"Golatkar"},{"article-title":"Learning multiple layers of features from tiny images","year":"2009","author":"Krizhevsky","key":"ref62"}],"container-title":["IEEE Access"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6287639\/9668973\/09766139.pdf?arnumber=9766139","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,1,22]],"date-time":"2024-01-22T22:38:29Z","timestamp":1705963109000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9766139\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022]]},"references-count":62,"URL":"https:\/\/doi.org\/10.1109\/access.2022.3171910","relation":{},"ISSN":["2169-3536"],"issn-type":[{"type":"electronic","value":"2169-3536"}],"subject":[],"published":{"date-parts":[[2022]]}}}