{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,5]],"date-time":"2026-08-05T18:14:46Z","timestamp":1785953686839,"version":"3.56.0"},"publisher-location":"New York, NY, USA","reference-count":42,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,10,21]],"date-time":"2022-10-21T00:00:00Z","timestamp":1666310400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,10,21]]},"DOI":"10.1145\/3571560.3571566","type":"proceedings-article","created":{"date-parts":[[2023,1,14]],"date-time":"2023-01-14T22:41:00Z","timestamp":1673736060000},"page":"40-49","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":22,"title":["Batch Layer Normalization A new normalization layer for CNNs and RNNs"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0648-489X","authenticated-orcid":false,"given":"Amir","family":"Ziaee","sequence":"first","affiliation":[{"name":"Design Computing Group, TU Wien, Austria"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5496-3860","authenticated-orcid":false,"given":"Erion","family":"\u00c7Ano","sequence":"additional","affiliation":[{"name":"Research Group Data Mining and Machine Learning, University of Vienna, Austria"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2023,1,12]]},"reference":[{"key":"#cr-split#-e_1_3_2_1_1_1.1","doi-asserted-by":"crossref","unstructured":"Rami Al-Rfou Dokook Choe Noah Constant Mandy Guo and Llion Jones. 2019. Character-Level Language Modeling with Deeper Self-Attention. In Proceedings of the Thirty-Third AAAI Conference on Artificial Intelligence and Thirty-First Innovative Applications of Artificial Intelligence Conference and Ninth AAAI Symposium on Educational Advances in Artificial Intelligence (Honolulu Hawaii USA) (AAAI'19\/IAAI'19\/EAAI'19). AAAI Press Article 388 8\u00a0pages. https:\/\/doi.org\/10.1609\/aaai.v33i01.33013159 10.1609\/aaai.v33i01.33013159","DOI":"10.1609\/aaai.v33i01.33013159"},{"key":"#cr-split#-e_1_3_2_1_1_1.2","doi-asserted-by":"crossref","unstructured":"Rami Al-Rfou Dokook Choe Noah Constant Mandy Guo and Llion Jones. 2019. Character-Level Language Modeling with Deeper Self-Attention. In Proceedings of the Thirty-Third AAAI Conference on Artificial Intelligence and Thirty-First Innovative Applications of Artificial Intelligence Conference and Ninth AAAI Symposium on Educational Advances in Artificial Intelligence (Honolulu Hawaii USA) (AAAI'19\/IAAI'19\/EAAI'19). AAAI Press Article 388 8\u00a0pages. https:\/\/doi.org\/10.1609\/aaai.v33i01.33013159","DOI":"10.1609\/aaai.v33i01.33013159"},{"key":"e_1_3_2_1_2_1","volume-title":"Proceedings of The 33rd International Conference on Machine Learning(Proceedings of Machine Learning Research, Vol.\u00a048)","author":"Arpit Devansh","year":"2016","unstructured":"Devansh Arpit , Yingbo Zhou , Bhargava Kota , and Venu Govindaraju . 2016 . Normalization Propagation: A Parametric Technique for Removing Internal Covariate Shift in Deep Networks . In Proceedings of The 33rd International Conference on Machine Learning(Proceedings of Machine Learning Research, Vol.\u00a048) , Maria\u00a0Florina Balcan and Kilian\u00a0Q. Weinberger (Eds.). PMLR, New York, New York, USA, 1168\u20131176. https:\/\/proceedings.mlr.press\/v48\/arpitb16.html Devansh Arpit, Yingbo Zhou, Bhargava Kota, and Venu Govindaraju. 2016. Normalization Propagation: A Parametric Technique for Removing Internal Covariate Shift in Deep Networks. In Proceedings of The 33rd International Conference on Machine Learning(Proceedings of Machine Learning Research, Vol.\u00a048), Maria\u00a0Florina Balcan and Kilian\u00a0Q. Weinberger (Eds.). PMLR, New York, New York, USA, 1168\u20131176. https:\/\/proceedings.mlr.press\/v48\/arpitb16.html"},{"key":"e_1_3_2_1_3_1","unstructured":"Lei\u00a0Jimmy Ba Jamie\u00a0Ryan Kiros and Geoffrey\u00a0E. Hinton. 2016. Layer Normalization. CoRR abs\/1607.06450(2016). arXiv:1607.06450http:\/\/arxiv.org\/abs\/1607.06450  Lei\u00a0Jimmy Ba Jamie\u00a0Ryan Kiros and Geoffrey\u00a0E. Hinton. 2016. Layer Normalization. CoRR abs\/1607.06450(2016). arXiv:1607.06450http:\/\/arxiv.org\/abs\/1607.06450"},{"key":"e_1_3_2_1_4_1","unstructured":"Aditya Bhatt Max Argus Artemij Amiranashvili and Thomas Brox. 2019. CrossNorm: Normalization for Off-Policy TD Reinforcement Learning. CoRR abs\/1902.05605(2019). arXiv:1902.05605http:\/\/arxiv.org\/abs\/1902.05605  Aditya Bhatt Max Argus Artemij Amiranashvili and Thomas Brox. 2019. CrossNorm: Normalization for Off-Policy TD Reinforcement Learning. CoRR abs\/1902.05605(2019). arXiv:1902.05605http:\/\/arxiv.org\/abs\/1902.05605"},{"key":"e_1_3_2_1_5_1","unstructured":"Angel\u00a0X. Chang Thomas\u00a0A. Funkhouser Leonidas\u00a0J. Guibas Pat Hanrahan Qi-Xing Huang Zimo Li Silvio Savarese Manolis Savva Shuran Song Hao Su Jianxiong Xiao Li Yi and Fisher Yu. 2015. ShapeNet: An Information-Rich 3D Model Repository. CoRR abs\/1512.03012(2015). arXiv:1512.03012http:\/\/arxiv.org\/abs\/1512.03012  Angel\u00a0X. Chang Thomas\u00a0A. Funkhouser Leonidas\u00a0J. Guibas Pat Hanrahan Qi-Xing Huang Zimo Li Silvio Savarese Manolis Savva Shuran Song Hao Su Jianxiong Xiao Li Yi and Fisher Yu. 2015. ShapeNet: An Information-Rich 3D Model Repository. CoRR abs\/1512.03012(2015). arXiv:1512.03012http:\/\/arxiv.org\/abs\/1512.03012"},{"key":"e_1_3_2_1_6_1","volume-title":"5th International Conference on Learning Representations, ICLR","author":"Cooijmans Tim","year":"2017","unstructured":"Tim Cooijmans , Nicolas Ballas , C\u00e9sar Laurent , \u00c7aglar G\u00fcl\u00e7ehre , and Aaron\u00a0 C. Courville . 2017. Recurrent Batch Normalization . In 5th International Conference on Learning Representations, ICLR 2017 , Toulon, France, April 24-26, 2017, Conference Track Proceedings. OpenReview .net. https:\/\/openreview.net\/forum?id=r1VdcHcxx Tim Cooijmans, Nicolas Ballas, C\u00e9sar Laurent, \u00c7aglar G\u00fcl\u00e7ehre, and Aaron\u00a0C. Courville. 2017. Recurrent Batch Normalization. In 5th International Conference on Learning Representations, ICLR 2017, Toulon, France, April 24-26, 2017, Conference Track Proceedings. OpenReview.net. https:\/\/openreview.net\/forum?id=r1VdcHcxx"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-48308-5_54"},{"key":"e_1_3_2_1_8_1","unstructured":"Igor Gitman and Boris Ginsburg. 2017. Comparison of Batch Normalization and Weight Normalization Algorithms for the Large-scale Image Classification. CoRR abs\/1709.08145(2017). arXiv:1709.08145http:\/\/arxiv.org\/abs\/1709.08145  Igor Gitman and Boris Ginsburg. 2017. Comparison of Batch Normalization and Weight Normalization Algorithms for the Large-scale Image Classification. CoRR abs\/1709.08145(2017). arXiv:1709.08145http:\/\/arxiv.org\/abs\/1709.08145"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46493-0_38"},{"key":"e_1_3_2_1_10_1","volume-title":"Advances in Neural Information Processing Systems 30: Annual Conference on Neural Information Processing Systems 2017","author":"Hoffer Elad","year":"2017","unstructured":"Elad Hoffer , Itay Hubara , and Daniel Soudry . 2017 . Train longer, generalize better: closing the generalization gap in large batch training of neural networks . In Advances in Neural Information Processing Systems 30: Annual Conference on Neural Information Processing Systems 2017 , December 4-9, 2017, Long Beach, CA, USA, Isabelle Guyon, Ulrike von Luxburg, Samy Bengio, Hanna\u00a0M. Wallach, Rob Fergus, S.\u00a0V.\u00a0N. Vishwanathan, and Roman Garnett (Eds.). 1731\u20131741. https:\/\/proceedings.neurips.cc\/paper\/ 2017\/hash\/a5e0ff62be0b08456fc7f1e88812af3d-Abstract.html Elad Hoffer, Itay Hubara, and Daniel Soudry. 2017. Train longer, generalize better: closing the generalization gap in large batch training of neural networks. In Advances in Neural Information Processing Systems 30: Annual Conference on Neural Information Processing Systems 2017, December 4-9, 2017, Long Beach, CA, USA, Isabelle Guyon, Ulrike von Luxburg, Samy Bengio, Hanna\u00a0M. Wallach, Rob Fergus, S.\u00a0V.\u00a0N. Vishwanathan, and Roman Garnett (Eds.). 1731\u20131741. https:\/\/proceedings.neurips.cc\/paper\/2017\/hash\/a5e0ff62be0b08456fc7f1e88812af3d-Abstract.html"},{"key":"e_1_3_2_1_11_1","volume-title":"Densely Connected Convolutional Networks. In 2017 IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2017","author":"Huang Gao","year":"2017","unstructured":"Gao Huang , Zhuang Liu , Laurens van\u00a0der Maaten , and Kilian\u00a0 Q. Weinberger . 2017 . Densely Connected Convolutional Networks. In 2017 IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2017 , Honolulu, HI, USA , July 21-26, 2017. IEEE Computer Society, 2261\u20132269. https:\/\/doi.org\/10.1109\/CVPR.2017.243 10.1109\/CVPR.2017.243 Gao Huang, Zhuang Liu, Laurens van\u00a0der Maaten, and Kilian\u00a0Q. Weinberger. 2017. Densely Connected Convolutional Networks. In 2017 IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2017, Honolulu, HI, USA, July 21-26, 2017. IEEE Computer Society, 2261\u20132269. https:\/\/doi.org\/10.1109\/CVPR.2017.243"},{"key":"e_1_3_2_1_12_1","unstructured":"Sergey Ioffe. 2017. Batch Renormalization: Towards Reducing Minibatch Dependence in Batch-Normalized Models. In Advances in Neural Information Processing Systems I.\u00a0Guyon U.\u00a0Von Luxburg S.\u00a0Bengio H.\u00a0Wallach R.\u00a0Fergus S.\u00a0Vishwanathan and R.\u00a0Garnett (Eds.). Vol.\u00a030. Curran Associates Inc.https:\/\/proceedings.neurips.cc\/paper\/2017\/file\/c54e7837e0cd0ced286cb5995327d1ab-Paper.pdf  Sergey Ioffe. 2017. Batch Renormalization: Towards Reducing Minibatch Dependence in Batch-Normalized Models. In Advances in Neural Information Processing Systems I.\u00a0Guyon U.\u00a0Von Luxburg S.\u00a0Bengio H.\u00a0Wallach R.\u00a0Fergus S.\u00a0Vishwanathan and R.\u00a0Garnett (Eds.). Vol.\u00a030. Curran Associates Inc.https:\/\/proceedings.neurips.cc\/paper\/2017\/file\/c54e7837e0cd0ced286cb5995327d1ab-Paper.pdf"},{"key":"e_1_3_2_1_13_1","volume-title":"Proceedings of the 32nd International Conference on Machine Learning(Proceedings of Machine Learning Research, Vol.\u00a037)","author":"Ioffe Sergey","year":"2015","unstructured":"Sergey Ioffe and Christian Szegedy . 2015 . Batch Normalization: Accelerating Deep Network Training by Reducing Internal Covariate Shift . In Proceedings of the 32nd International Conference on Machine Learning(Proceedings of Machine Learning Research, Vol.\u00a037) , Francis Bach and David Blei (Eds.). PMLR, Lille, France, 448\u2013456. https:\/\/proceedings.mlr.press\/v37\/ioffe15.html Sergey Ioffe and Christian Szegedy. 2015. Batch Normalization: Accelerating Deep Network Training by Reducing Internal Covariate Shift. In Proceedings of the 32nd International Conference on Machine Learning(Proceedings of Machine Learning Research, Vol.\u00a037), Francis Bach and David Blei (Eds.). PMLR, Lille, France, 448\u2013456. https:\/\/proceedings.mlr.press\/v37\/ioffe15.html"},{"key":"e_1_3_2_1_14_1","unstructured":"Alex Krizhevsky Vinod Nair and Geoffrey Hinton. 2009. CIFAR-10 (Canadian Institute for Advanced Research). (2009). http:\/\/www.cs.toronto.edu\/\u00a0kriz\/cifar.html  Alex Krizhevsky Vinod Nair and Geoffrey Hinton. 2009. CIFAR-10 (Canadian Institute for Advanced Research). (2009). http:\/\/www.cs.toronto.edu\/\u00a0kriz\/cifar.html"},{"key":"e_1_3_2_1_15_1","volume-title":"Proceedings of the 36th International Conference on Machine Learning(Proceedings of Machine Learning Research, Vol.\u00a097)","author":"Kurach Karol","year":"2019","unstructured":"Karol Kurach , Mario Lu\u010di\u0107 , Xiaohua Zhai , Marcin Michalski , and Sylvain Gelly . 2019 . A Large-Scale Study on Regularization and Normalization in GANs . In Proceedings of the 36th International Conference on Machine Learning(Proceedings of Machine Learning Research, Vol.\u00a097) , Kamalika Chaudhuri and Ruslan Salakhutdinov (Eds.). PMLR, 3581\u20133590. https:\/\/proceedings.mlr.press\/v97\/kurach19a.html Karol Kurach, Mario Lu\u010di\u0107, Xiaohua Zhai, Marcin Michalski, and Sylvain Gelly. 2019. A Large-Scale Study on Regularization and Normalization in GANs. In Proceedings of the 36th International Conference on Machine Learning(Proceedings of Machine Learning Research, Vol.\u00a097), Kamalika Chaudhuri and Ruslan Salakhutdinov (Eds.). PMLR, 3581\u20133590. https:\/\/proceedings.mlr.press\/v97\/kurach19a.html"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472159"},{"key":"e_1_3_2_1_17_1","volume-title":"Streaming Normalization: Towards Simpler and More Biologically-plausible Normalizations for Online and Recurrent Learning. CoRR abs\/1610.06160(2016). arXiv:1610.06160http:\/\/arxiv.org\/abs\/1610.06160","author":"Liao Qianli","year":"2016","unstructured":"Qianli Liao , Kenji Kawaguchi , and Tomaso\u00a0 A. Poggio . 2016 . Streaming Normalization: Towards Simpler and More Biologically-plausible Normalizations for Online and Recurrent Learning. CoRR abs\/1610.06160(2016). arXiv:1610.06160http:\/\/arxiv.org\/abs\/1610.06160 Qianli Liao, Kenji Kawaguchi, and Tomaso\u00a0A. Poggio. 2016. Streaming Normalization: Towards Simpler and More Biologically-plausible Normalizations for Online and Recurrent Learning. CoRR abs\/1610.06160(2016). arXiv:1610.06160http:\/\/arxiv.org\/abs\/1610.06160"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.cviu.2017.05.007"},{"key":"e_1_3_2_1_20_1","volume-title":"4th International Conference on Learning Representations, ICLR","author":"Neyshabur Behnam","year":"2016","unstructured":"Behnam Neyshabur , Ryota Tomioka , Ruslan Salakhutdinov , and Nathan Srebro . 2016. Data-Dependent Path Normalization in Neural Networks . In 4th International Conference on Learning Representations, ICLR 2016 , San Juan, Puerto Rico , May 2-4, 2016, Conference Track Proceedings, Yoshua Bengio and Yann LeCun (Eds .). http:\/\/arxiv.org\/abs\/1511.06747 Behnam Neyshabur, Ryota Tomioka, Ruslan Salakhutdinov, and Nathan Srebro. 2016. Data-Dependent Path Normalization in Neural Networks. In 4th International Conference on Learning Representations, ICLR 2016, San Juan, Puerto Rico, May 2-4, 2016, Conference Track Proceedings, Yoshua Bengio and Yann LeCun (Eds.). http:\/\/arxiv.org\/abs\/1511.06747"},{"key":"e_1_3_2_1_21_1","volume-title":"2017 IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2017","author":"Qi Charles\u00a0Ruizhongtai","year":"2017","unstructured":"Charles\u00a0Ruizhongtai Qi , Hao Su , Kaichun Mo , and Leonidas\u00a0 J. Guibas . 2017 . PointNet: Deep Learning on Point Sets for 3D Classification and Segmentation . In 2017 IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2017 , Honolulu, HI, USA , July 21-26, 2017. IEEE Computer Society, 77\u201385. https:\/\/doi.org\/10.1109\/CVPR.2017.16 10.1109\/CVPR.2017.16 Charles\u00a0Ruizhongtai Qi, Hao Su, Kaichun Mo, and Leonidas\u00a0J. Guibas. 2017. PointNet: Deep Learning on Point Sets for 3D Classification and Segmentation. In 2017 IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2017, Honolulu, HI, USA, July 21-26, 2017. IEEE Computer Society, 77\u201385. https:\/\/doi.org\/10.1109\/CVPR.2017.16"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-015-0816-y"},{"key":"e_1_3_2_1_23_1","volume-title":"Improved Techniques for Training GANs. In Advances in Neural Information Processing Systems 29: Annual Conference on Neural Information Processing Systems 2016","author":"Salimans Tim","year":"2016","unstructured":"Tim Salimans , Ian\u00a0 J. Goodfellow , Wojciech Zaremba , Vicki Cheung , Alec Radford , and Xi Chen . 2016 . Improved Techniques for Training GANs. In Advances in Neural Information Processing Systems 29: Annual Conference on Neural Information Processing Systems 2016 , December 5-10, 2016, Barcelona, Spain, Daniel\u00a0D. Lee, Masashi Sugiyama, Ulrike von Luxburg, Isabelle Guyon, and Roman Garnett (Eds.). 2226\u20132234. https:\/\/proceedings.neurips.cc\/paper\/ 2016\/hash\/8a3363abe792db2d8761d6403605aeb7-Abstract.html Tim Salimans, Ian\u00a0J. Goodfellow, Wojciech Zaremba, Vicki Cheung, Alec Radford, and Xi Chen. 2016. Improved Techniques for Training GANs. In Advances in Neural Information Processing Systems 29: Annual Conference on Neural Information Processing Systems 2016, December 5-10, 2016, Barcelona, Spain, Daniel\u00a0D. Lee, Masashi Sugiyama, Ulrike von Luxburg, Isabelle Guyon, and Roman Garnett (Eds.). 2226\u20132234. https:\/\/proceedings.neurips.cc\/paper\/2016\/hash\/8a3363abe792db2d8761d6403605aeb7-Abstract.html"},{"key":"e_1_3_2_1_24_1","volume-title":"Advances in Neural Information Processing Systems 29: Annual Conference on Neural Information Processing Systems 2016","author":"Salimans Tim","year":"2016","unstructured":"Tim Salimans and Diederik\u00a0 P. Kingma . 2016 . Weight Normalization: A Simple Reparameterization to Accelerate Training of Deep Neural Networks . In Advances in Neural Information Processing Systems 29: Annual Conference on Neural Information Processing Systems 2016 , December 5-10, 2016, Barcelona, Spain, Daniel\u00a0D. Lee, Masashi Sugiyama, Ulrike von Luxburg, Isabelle Guyon, and Roman Garnett (Eds.). 901. https:\/\/proceedings.neurips.cc\/paper\/ 2016\/hash\/ed265bc903a5a097f61d3ec064d96d2e-Abstract.html Tim Salimans and Diederik\u00a0P. Kingma. 2016. Weight Normalization: A Simple Reparameterization to Accelerate Training of Deep Neural Networks. In Advances in Neural Information Processing Systems 29: Annual Conference on Neural Information Processing Systems 2016, December 5-10, 2016, Barcelona, Spain, Daniel\u00a0D. Lee, Masashi Sugiyama, Ulrike von Luxburg, Isabelle Guyon, and Roman Garnett (Eds.). 901. https:\/\/proceedings.neurips.cc\/paper\/2016\/hash\/ed265bc903a5a097f61d3ec064d96d2e-Abstract.html"},{"key":"e_1_3_2_1_25_1","unstructured":"Sheng Shen Zhewei Yao Amir Gholami Michael\u00a0W. Mahoney and Kurt Keutzer. 2020. Rethinking Batch Normalization in Transformers. CoRR abs\/2003.07845(2020). arXiv:2003.07845https:\/\/arxiv.org\/abs\/2003.07845  Sheng Shen Zhewei Yao Amir Gholami Michael\u00a0W. Mahoney and Kurt Keutzer. 2020. Rethinking Batch Normalization in Transformers. CoRR abs\/2003.07845(2020). arXiv:2003.07845https:\/\/arxiv.org\/abs\/2003.07845"},{"key":"e_1_3_2_1_26_1","unstructured":"Robert Stojnic Ross Taylor Marcin Kardas Viktor Kerkez and Ludovic Viaud. 2022. Papers with Code-The latest in Machine Learning. https:\/\/paperswithcode.com\/method\/layer-normalization. Accessed: 2022-06-24.  Robert Stojnic Ross Taylor Marcin Kardas Viktor Kerkez and Ludovic Viaud. 2022. Papers with Code-The latest in Machine Learning. https:\/\/paperswithcode.com\/method\/layer-normalization. Accessed: 2022-06-24."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"e_1_3_2_1_28_1","volume-title":"Rethinking the Inception Architecture for Computer Vision. In 2016 IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2016","author":"Szegedy Christian","year":"2016","unstructured":"Christian Szegedy , Vincent Vanhoucke , Sergey Ioffe , Jonathon Shlens , and Zbigniew Wojna . 2016 . Rethinking the Inception Architecture for Computer Vision. In 2016 IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2016 , Las Vegas, NV, USA , June 27-30, 2016. IEEE Computer Society, 2818\u20132826. https:\/\/doi.org\/10.1109\/CVPR.2016.308 10.1109\/CVPR.2016.308 Christian Szegedy, Vincent Vanhoucke, Sergey Ioffe, Jonathon Shlens, and Zbigniew Wojna. 2016. Rethinking the Inception Architecture for Computer Vision. In 2016 IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2016, Las Vegas, NV, USA, June 27-30, 2016. IEEE Computer Society, 2818\u20132826. https:\/\/doi.org\/10.1109\/CVPR.2016.308"},{"key":"e_1_3_2_1_29_1","volume-title":"Instance Normalization: The Missing Ingredient for Fast Stylization. CoRR abs\/1607.08022(2016). arXiv:1607.08022http:\/\/arxiv.org\/abs\/1607.08022","author":"Ulyanov Dmitry","year":"2016","unstructured":"Dmitry Ulyanov , Andrea Vedaldi , and Victor\u00a0 S. Lempitsky . 2016 . Instance Normalization: The Missing Ingredient for Fast Stylization. CoRR abs\/1607.08022(2016). arXiv:1607.08022http:\/\/arxiv.org\/abs\/1607.08022 Dmitry Ulyanov, Andrea Vedaldi, and Victor\u00a0S. Lempitsky. 2016. Instance Normalization: The Missing Ingredient for Fast Stylization. CoRR abs\/1607.08022(2016). arXiv:1607.08022http:\/\/arxiv.org\/abs\/1607.08022"},{"key":"e_1_3_2_1_30_1","volume-title":"Advances in Neural Information Processing Systems 30: Annual Conference on Neural Information Processing Systems 2017","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani , Noam Shazeer , Niki Parmar , Jakob Uszkoreit , Llion Jones , Aidan\u00a0 N. Gomez , Lukasz Kaiser , and Illia Polosukhin . 2017 . Attention is All you Need . In Advances in Neural Information Processing Systems 30: Annual Conference on Neural Information Processing Systems 2017 , December 4-9, 2017, Long Beach, CA, USA, Isabelle Guyon, Ulrike von Luxburg, Samy Bengio, Hanna\u00a0M. Wallach, Rob Fergus, S.\u00a0V.\u00a0N. Vishwanathan, and Roman Garnett (Eds.). 5998\u20136008. https:\/\/proceedings.neurips.cc\/paper\/ 2017\/hash\/3f5ee243547dee91fbd053c1c4a845aa-Abstract.html Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan\u00a0N. Gomez, Lukasz Kaiser, and Illia Polosukhin. 2017. Attention is All you Need. In Advances in Neural Information Processing Systems 30: Annual Conference on Neural Information Processing Systems 2017, December 4-9, 2017, Long Beach, CA, USA, Isabelle Guyon, Ulrike von Luxburg, Samy Bengio, Hanna\u00a0M. Wallach, Rob Fergus, S.\u00a0V.\u00a0N. Vishwanathan, and Roman Garnett (Eds.). 5998\u20136008. https:\/\/proceedings.neurips.cc\/paper\/2017\/hash\/3f5ee243547dee91fbd053c1c4a845aa-Abstract.html"},{"key":"e_1_3_2_1_31_1","volume-title":"Advances in Neural Information Processing Systems, M.\u00a0Ranzato, A.\u00a0Beygelzimer, Y.\u00a0Dauphin, P.S. Liang, and J.\u00a0Wortman Vaughan (Eds.). Vol.\u00a034. Curran Associates","author":"Wan Ruosi","year":"2021","unstructured":"Ruosi Wan , Zhanxing Zhu , Xiangyu Zhang , and Jian Sun . 2021. Spherical Motion Dynamics: Learning Dynamics of Normalized Neural Network using SGD and Weight Decay . In Advances in Neural Information Processing Systems, M.\u00a0Ranzato, A.\u00a0Beygelzimer, Y.\u00a0Dauphin, P.S. Liang, and J.\u00a0Wortman Vaughan (Eds.). Vol.\u00a034. Curran Associates , Inc ., 6380\u20136391. https:\/\/proceedings.neurips.cc\/paper\/ 2021 \/file\/326a8c055c0d04f5b06544665d8bb3ea-Paper.pdf Ruosi Wan, Zhanxing Zhu, Xiangyu Zhang, and Jian Sun. 2021. Spherical Motion Dynamics: Learning Dynamics of Normalized Neural Network using SGD and Weight Decay. In Advances in Neural Information Processing Systems, M.\u00a0Ranzato, A.\u00a0Beygelzimer, Y.\u00a0Dauphin, P.S. Liang, and J.\u00a0Wortman Vaughan (Eds.). Vol.\u00a034. Curran Associates, Inc., 6380\u20136391. https:\/\/proceedings.neurips.cc\/paper\/2021\/file\/326a8c055c0d04f5b06544665d8bb3ea-Paper.pdf"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01261-8_1"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.634"},{"key":"e_1_3_2_1_34_1","volume-title":"Proceedings of the 37th International Conference on Machine Learning, ICML 2020","author":"Xiong Ruibin","year":"2020","unstructured":"Ruibin Xiong , Yunchang Yang , Di He , Kai Zheng , Shuxin Zheng , Chen Xing , Huishuai Zhang , Yanyan Lan , Liwei Wang , and Tie-Yan Liu . 2020 . On Layer Normalization in the Transformer Architecture . In Proceedings of the 37th International Conference on Machine Learning, ICML 2020 , 13-18 July 2020, Virtual Event(Proceedings of Machine Learning Research, Vol.\u00a0119). PMLR, 10524\u201310533. http:\/\/proceedings.mlr.press\/v119\/xiong20b.html Ruibin Xiong, Yunchang Yang, Di He, Kai Zheng, Shuxin Zheng, Chen Xing, Huishuai Zhang, Yanyan Lan, Liwei Wang, and Tie-Yan Liu. 2020. On Layer Normalization in the Transformer Architecture. In Proceedings of the 37th International Conference on Machine Learning, ICML 2020, 13-18 July 2020, Virtual Event(Proceedings of Machine Learning Research, Vol.\u00a0119). PMLR, 10524\u201310533. http:\/\/proceedings.mlr.press\/v119\/xiong20b.html"},{"key":"e_1_3_2_1_35_1","volume-title":"Understanding and Improving Layer Normalization. In Advances in Neural Information Processing Systems 32: Annual Conference on Neural Information Processing Systems 2019","author":"Xu Jingjing","year":"2019","unstructured":"Jingjing Xu , Xu Sun , Zhiyuan Zhang , Guangxiang Zhao , and Junyang Lin . 2019 . Understanding and Improving Layer Normalization. In Advances in Neural Information Processing Systems 32: Annual Conference on Neural Information Processing Systems 2019 , NeurIPS 2019, December 8-14, 2019, Vancouver, BC, Canada, Hanna\u00a0M. Wallach, Hugo Larochelle, Alina Beygelzimer, Florence d\u2019Alch\u00e9-Buc, Emily\u00a0B. Fox, and Roman Garnett (Eds.). 4383\u20134393. https:\/\/proceedings.neurips.cc\/paper\/2019\/hash\/2f4fe03d77724a7217006e5d16728874-Abstract.html Jingjing Xu, Xu Sun, Zhiyuan Zhang, Guangxiang Zhao, and Junyang Lin. 2019. Understanding and Improving Layer Normalization. In Advances in Neural Information Processing Systems 32: Annual Conference on Neural Information Processing Systems 2019, NeurIPS 2019, December 8-14, 2019, Vancouver, BC, Canada, Hanna\u00a0M. Wallach, Hugo Larochelle, Alina Beygelzimer, Florence d\u2019Alch\u00e9-Buc, Emily\u00a0B. Fox, and Roman Garnett (Eds.). 4383\u20134393. https:\/\/proceedings.neurips.cc\/paper\/2019\/hash\/2f4fe03d77724a7217006e5d16728874-Abstract.html"},{"key":"e_1_3_2_1_36_1","volume-title":"XLNet: Generalized Autoregressive Pretraining for Language Understanding. In Advances in Neural Information Processing Systems 32: Annual Conference on Neural Information Processing Systems 2019","author":"Yang Zhilin","year":"2019","unstructured":"Zhilin Yang , Zihang Dai , Yiming Yang , Jaime\u00a0 G. Carbonell , Ruslan Salakhutdinov , and Quoc\u00a0 V. Le . 2019 . XLNet: Generalized Autoregressive Pretraining for Language Understanding. In Advances in Neural Information Processing Systems 32: Annual Conference on Neural Information Processing Systems 2019 , NeurIPS 2019, December 8-14, 2019, Vancouver, BC, Canada, Hanna\u00a0M. Wallach, Hugo Larochelle, Alina Beygelzimer, Florence d\u2019Alch\u00e9-Buc, Emily\u00a0B. Fox, and Roman Garnett (Eds.). 5754\u20135764. https:\/\/proceedings.neurips.cc\/paper\/2019\/hash\/dc6a7e655d7e5840e66733e9ee67cc69-Abstract.html Zhilin Yang, Zihang Dai, Yiming Yang, Jaime\u00a0G. Carbonell, Ruslan Salakhutdinov, and Quoc\u00a0V. Le. 2019. XLNet: Generalized Autoregressive Pretraining for Language Understanding. In Advances in Neural Information Processing Systems 32: Annual Conference on Neural Information Processing Systems 2019, NeurIPS 2019, December 8-14, 2019, Vancouver, BC, Canada, Hanna\u00a0M. Wallach, Hugo Larochelle, Alina Beygelzimer, Florence d\u2019Alch\u00e9-Buc, Emily\u00a0B. Fox, and Roman Garnett (Eds.). 5754\u20135764. https:\/\/proceedings.neurips.cc\/paper\/2019\/hash\/dc6a7e655d7e5840e66733e9ee67cc69-Abstract.html"},{"key":"e_1_3_2_1_37_1","volume-title":"Gradient Centralization: A New Optimization Technique for Deep Neural Networks. In Computer Vision - ECCV 2020 - 16th European Conference","author":"Yong Hongwei","year":"2020","unstructured":"Hongwei Yong , Jianqiang Huang , Xiansheng Hua , and Lei Zhang . 2020 . Gradient Centralization: A New Optimization Technique for Deep Neural Networks. In Computer Vision - ECCV 2020 - 16th European Conference , Glasgow, UK , August 23-28, 2020, Proceedings, Part I(Lecture Notes in Computer Science, Vol.\u00a012346), Andrea Vedaldi, Horst Bischof, Thomas Brox, and Jan-Michael Frahm (Eds.). Springer , 635\u2013652. https:\/\/doi.org\/10.1007\/978-3-030-58452-8_37 10.1007\/978-3-030-58452-8_37 Hongwei Yong, Jianqiang Huang, Xiansheng Hua, and Lei Zhang. 2020. Gradient Centralization: A New Optimization Technique for Deep Neural Networks. In Computer Vision - ECCV 2020 - 16th European Conference, Glasgow, UK, August 23-28, 2020, Proceedings, Part I(Lecture Notes in Computer Science, Vol.\u00a012346), Andrea Vedaldi, Horst Bischof, Thomas Brox, and Jan-Michael Frahm (Eds.). Springer, 635\u2013652. https:\/\/doi.org\/10.1007\/978-3-030-58452-8_37"},{"key":"e_1_3_2_1_38_1","volume-title":"6th International Conference on Learning Representations, ICLR","author":"Yu Adams\u00a0Wei","year":"2018","unstructured":"Adams\u00a0Wei Yu , David Dohan , Minh-Thang Luong , Rui Zhao , Kai Chen , Mohammad Norouzi , and Quoc\u00a0 V. Le. 2018. QANet: Combining Local Convolution with Global Self-Attention for Reading Comprehension . In 6th International Conference on Learning Representations, ICLR 2018 , Vancouver, BC , Canada, April 30 - May 3, 2018, Conference Track Proceedings. OpenReview .net. https:\/\/openreview.net\/forum?id=B14TlG-RW Adams\u00a0Wei Yu, David Dohan, Minh-Thang Luong, Rui Zhao, Kai Chen, Mohammad Norouzi, and Quoc\u00a0V. Le. 2018. QANet: Combining Local Convolution with Global Self-Attention for Reading Comprehension. In 6th International Conference on Learning Representations, ICLR 2018, Vancouver, BC, Canada, April 30 - May 3, 2018, Conference Track Proceedings. OpenReview.net. https:\/\/openreview.net\/forum?id=B14TlG-RW"},{"key":"#cr-split#-e_1_3_2_1_39_1.1","unstructured":"Tong Yu and Hong Zhu. 2020. Hyper-Parameter Optimization: A Review of Algorithms and Applications. CoRR abs\/2003.05689(2020). arXiv:2003.05689https:\/\/doi.org\/10.48550\/arXiv.2003.05689 10.48550\/arXiv.2003.05689"},{"key":"#cr-split#-e_1_3_2_1_39_1.2","unstructured":"Tong Yu and Hong Zhu. 2020. Hyper-Parameter Optimization: A Review of Algorithms and Applications. CoRR abs\/2003.05689(2020). arXiv:2003.05689https:\/\/doi.org\/10.48550\/arXiv.2003.05689"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.5244\/C.30.87"}],"event":{"name":"ICAAI 2022: 2022 The 6th International Conference on Advances in Artificial Intelligence","location":"Birmingham United Kingdom","acronym":"ICAAI 2022"},"container-title":["2022 The 6th International Conference on Advances in Artificial Intelligence"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3571560.3571566","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3571560.3571566","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T17:50:55Z","timestamp":1750182655000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3571560.3571566"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,10,21]]},"references-count":42,"alternative-id":["10.1145\/3571560.3571566","10.1145\/3571560"],"URL":"https:\/\/doi.org\/10.1145\/3571560.3571566","relation":{},"subject":[],"published":{"date-parts":[[2022,10,21]]},"assertion":[{"value":"2023-01-12","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}