{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,15]],"date-time":"2025-11-15T07:13:20Z","timestamp":1763190800223,"version":"3.45.0"},"reference-count":52,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,6,30]],"date-time":"2025-06-30T00:00:00Z","timestamp":1751241600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,6,30]],"date-time":"2025-06-30T00:00:00Z","timestamp":1751241600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,6,30]]},"DOI":"10.1109\/ijcnn64981.2025.11227594","type":"proceedings-article","created":{"date-parts":[[2025,11,14]],"date-time":"2025-11-14T18:46:15Z","timestamp":1763145975000},"page":"1-8","source":"Crossref","is-referenced-by-count":0,"title":["Multiplicative Stochastic Gradient Descent for fast and robust deep learning training"],"prefix":"10.1109","author":[{"given":"Manos","family":"Kirtas","sequence":"first","affiliation":[{"name":"Aristotle University of Thessaloniki,Computational Intelligence and Deep Learning Group,Dept. of Informatics,Thessaloniki,Greece"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Nikolaos","family":"Passalis","sequence":"additional","affiliation":[{"name":"Aristotle University of Thessaloniki,Computational Intelligence and Deep Learning Group,Dept. of Chemical Engineering,Thessaloniki,Greece"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Anastasios","family":"Tefas","sequence":"additional","affiliation":[{"name":"Aristotle University of Thessaloniki,Computational Intelligence and Deep Learning Group,Dept. of Informatics,Thessaloniki,Greece"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1088\/1742-5468\/ab39d9"},{"article-title":"High-performance large-scale image recognition without normalization","year":"2021","author":"Brock","key":"ref2"},{"key":"ref3","first-page":"21285","article-title":"Towards theoretically understanding why sgd generalizes better than adam in deep learning","volume":"33","author":"Zhou","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref4","first-page":"560","article-title":"signsgd: Compressed optimisation for non-convex problems","volume-title":"International Conference on Machine Learning","author":"Bernstein"},{"issue":"7","key":"ref5","article-title":"Adaptive subgradient methods for online learning and stochastic optimization","volume":"12","author":"Duchi","year":"2011","journal-title":"Journal of machine learning research"},{"article-title":"Adam: A method for stochastic optimization","year":"2014","author":"Kingma","key":"ref6"},{"article-title":"On the variance of the adaptive learning rate and beyond","year":"2019","author":"Liu","key":"ref7"},{"article-title":"On the convergence of adam and beyond","year":"2019","author":"Reddi","key":"ref8"},{"article-title":"Large batch training of convolutional networks","year":"2017","author":"You","key":"ref9"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.123"},{"key":"ref11","first-page":"249","article-title":"Understanding the difficulty of training deep feedforward neural networks","volume-title":"Proceedings of the thirteenth international conference on artificial intelligence and statistics. JMLR Workshop and Conference Proceedings","author":"Glorot"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/TETCI.2022.3182765"},{"key":"ref13","first-page":"21370","article-title":"On the distance between two neural networks and the stability of learning","volume-title":"Advances in Neural Information Processing Systems","volume":"33","author":"Bernstein","year":"2020"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/tcyb.2019.2950779"},{"issue":"6","key":"ref15","first-page":"121","article-title":"The multiplicative weights update method: a meta-algorithm and applications","volume-title":"Theory of Computing","volume":"8","author":"Arora","year":"2012"},{"key":"ref16","first-page":"206","article-title":"Adaptive multiplicative updates for quadratic nonnegative matrix factorization","volume-title":"Neurocomputing","volume":"134","author":"Zhang","year":"2014"},{"issue":"1","key":"ref17","first-page":"325","article-title":"The perceptron algorithm versus winnow: linear versus logarithmic mistake bounds when few input variables are relevant","volume-title":"Artificial Intelligence","volume":"97","author":"Kivinen","year":"1997"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1023\/A:1022869011914"},{"key":"ref19","article-title":"Algorithms for non-negative matrix factorization","volume-title":"Advances in Neural Information Processing Systems","volume":"13","author":"Lee","year":"2000"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1007\/s11590-019-01434-9"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1037\/h0042519"},{"key":"ref22","first-page":"6748","article-title":"Learning by turning: Neural architecture aware optimisation","volume-title":"Proceedings of the 38th International Conference on Machine Learning","volume":"139","author":"Liu"},{"key":"ref23","first-page":"13319","article-title":"Learning compositional functions via multiplicative weight updates","volume-title":"Advances in Neural Information Processing Systems","volume":"33","author":"Bernstein","year":"2020"},{"key":"ref24","first-page":"1352","article-title":"ReZero is All You Need: Fast Convergence at Large Depth","volume-title":"37th Conference on Uncertainty in Artificial Intelligence, UAI 2021","author":"Bachlechner"},{"article-title":"Why gradient clipping accelerates training: A theoretical justification for adaptivity","year":"2019","author":"Zhang","key":"ref25"},{"article-title":"Large batch training of convolutional networks","year":"2017","author":"You","key":"ref26"},{"article-title":"On the variance of the adaptive learning rate and beyond","year":"2019","author":"Liu","key":"ref27"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2017.2747861"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2014.2310059"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/ECOC52684.2021.9605987"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1515\/nanoph-2022-0423"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.123"},{"key":"ref33","first-page":"244","article-title":"On the optimization of deep networks: Implicit acceleration by overparameterization","volume-title":"Proceedings of the 35th International Conference on Machine Learning","volume":"80","author":"Arora"},{"article-title":"Qualitatively characterizing neural network optimization problems","year":"2014","author":"Goodfellow","key":"ref34"},{"article-title":"Understanding overparameterization in generative adversarial networks","year":"2021","author":"Balaji","key":"ref35"},{"article-title":"Automatic differentiation in pytorch","year":"2017","author":"Paszke","key":"ref36"},{"article-title":"TensorFlow: Large-scale machine learning on heterogeneous systems","year":"2015","author":"Abadi","key":"ref37"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2015.2479223"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1093\/comjnl\/3.3.175"},{"article-title":"Learning multiple layers of features from tiny images","year":"2009","author":"Krizhevsky","key":"ref40"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"article-title":"Very deep convolutional networks for large-scale image recognition","year":"2014","author":"Simonyan","key":"ref42"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.305"},{"article-title":"Micro-batch training with batch-channel normalization and weight standardization","year":"2019","author":"Qiao","key":"ref45"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1117\/1.AP.5.1.016004"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-74627-7_35"},{"key":"ref48","first-page":"194","article-title":"Online learning and generalization of parts-based image representations by non-negative sparse autoencoders","volume-title":"Neural Networks","volume":"33","author":"Lemme","year":"2012"},{"article-title":"Non-negative isomorphic neural networks for photonic neuromorphic accelerators","year":"2023","author":"Kirtas","key":"ref49"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN.2018.8489216"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1088\/2634-4386\/ac724d"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/JSTQE.2023.3277118"}],"event":{"name":"2025 International Joint Conference on Neural Networks (IJCNN)","start":{"date-parts":[[2025,6,30]]},"location":"Rome, Italy","end":{"date-parts":[[2025,7,5]]}},"container-title":["2025 International Joint Conference on Neural Networks (IJCNN)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11227166\/11227148\/11227594.pdf?arnumber=11227594","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,15]],"date-time":"2025-11-15T07:10:31Z","timestamp":1763190631000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11227594\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,30]]},"references-count":52,"URL":"https:\/\/doi.org\/10.1109\/ijcnn64981.2025.11227594","relation":{},"subject":[],"published":{"date-parts":[[2025,6,30]]}}}