{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,2]],"date-time":"2026-08-02T06:45:57Z","timestamp":1785653157682,"version":"3.56.0"},"publisher-location":"Cham","reference-count":41,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032316653","type":"print"},{"value":"9783032316660","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,8,3]],"date-time":"2026-08-03T00:00:00Z","timestamp":1785715200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,8,3]],"date-time":"2026-08-03T00:00:00Z","timestamp":1785715200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-3-032-31666-0_12","type":"book-chapter","created":{"date-parts":[[2026,8,2]],"date-time":"2026-08-02T05:46:05Z","timestamp":1785649565000},"page":"175-190","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["EMASAM: a\u00a0Computationally Efficient Sharpness-Aware Minimization via\u00a0EMA-Guided Perturbations"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-5498-8301","authenticated-orcid":false,"given":"Tanapat","family":"Ratchatorn","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5756-1904","authenticated-orcid":false,"given":"Masayuki","family":"Tanaka","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,8,3]]},"reference":[{"key":"12_CR1","unstructured":"Becker, M., Altrock, F., Risse, B.: Momentum-sam: sharpness aware minimization without computational overhead (2024). arXiv:2401.12033 arXiv preprint"},{"issue":"12","key":"12_CR2","doi-asserted-by":"publisher","DOI":"10.1088\/1742-5468\/ab39d9","volume":"2019","author":"P Chaudhari","year":"2019","unstructured":"Chaudhari, P., et al.: Entropy-SGD: biasing gradient descent into wide valleys. J. Stat. Mech: Theory Exp. 2019(12), 124018 (2019)","journal-title":"J. Stat. Mech: Theory Exp."},{"key":"12_CR3","doi-asserted-by":"crossref","unstructured":"Chaudhari, P., Soatto, S.: Stochastic gradient descent performs variational inference, converges to limit cycles for deep networks. In: International Conference on Learning Representations (2018)","DOI":"10.1109\/ITA.2018.8503224"},{"key":"12_CR4","unstructured":"Chen, J., Gu, Q.: Closing the generalization gap of adaptive gradient methods in training deep neural networks. CoRR abs\/1806.06763 (2018). http:\/\/arxiv.org\/abs\/1806.06763"},{"key":"12_CR5","doi-asserted-by":"crossref","unstructured":"Cohen, G., Afshar, S., Tapson, J., Van Schaik, A.: Emnist: extending MNIST to handwritten letters. In: 2017 International Joint Conference on Neural Networks (IJCNN), pp. 2921\u20132926. IEEE (2017)","DOI":"10.1109\/IJCNN.2017.7966217"},{"key":"12_CR6","doi-asserted-by":"publisher","first-page":"9974","DOI":"10.52202\/079017-0320","volume":"37","author":"A Defazio","year":"2024","unstructured":"Defazio, A., Yang, X., Mehta, H., Mishchenko, K., Khaled, A., Cutkosky, A.: The road less scheduled. Adv. Neural. Inf. Process. Syst. 37, 9974\u201310007 (2024)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"12_CR7","doi-asserted-by":"publisher","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.J., Li, K., Fei-Fei, L.: Imagenet: a large-scale hierarchical image database. In: 2009 IEEE Conference on Computer Vision and Pattern Recognition, pp. 248\u2013255 (2009). https:\/\/doi.org\/10.1109\/CVPR.2009.5206848","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"12_CR8","unstructured":"DeVries, T., Taylor, G.W.: Improved regularization of convolutional neural networks with cutout. arXiv preprint arXiv:1708.04552 (2017)"},{"key":"12_CR9","unstructured":"Dosovitskiy, A.: An image is worth 16x16 words: transformers for image recognition at scale (2020). arXiv:2010.11929 arXiv preprint"},{"key":"12_CR10","unstructured":"Dziugaite, G.K., Roy, D.M.: Computing nonvacuous generalization bounds for deep (stochastic) neural networks with many more parameters than training data (2017). arXiv:1703.11008 arXiv preprint"},{"key":"12_CR11","unstructured":"Foret, P., Kleiner, A., Mobahi, H., Neyshabur, B.: Sharpness-aware minimization for efficiently improving generalization (2020). arXiv:2010.01412 arXiv preprint"},{"key":"12_CR12","unstructured":"Goyal, P., et al.: Accurate, large minibatch SGD: training imagenet in 1 hour. CoRR abs\/1706.02677 (2017). http:\/\/arxiv.org\/abs\/1706.02677"},{"key":"12_CR13","doi-asserted-by":"crossref","unstructured":"Han, D., Kim, J., Kim, J.: Deep pyramidal residual networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 5927\u20135935 (2017)","DOI":"10.1109\/CVPR.2017.668"},{"key":"12_CR14","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"issue":"1","key":"12_CR15","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1162\/neco.1997.9.1.1","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter, S., Schmidhuber, J.: Flat minima. Neural Comput. 9(1), 1\u201342 (1997)","journal-title":"Neural Comput."},{"key":"12_CR16","unstructured":"Jiang, W., Yang, H., Zhang, Y., Kwok, J.: An adaptive policy to employ sharpness-aware minimization. In: The Eleventh International Conference on Learning Representations (2023)"},{"key":"12_CR17","unstructured":"Keskar, N.S., Mudigere, D., Nocedal, J., Smelyanskiy, M., Tang, P.T.P.: On large-batch training for deep learning: generalization gap and sharp minima (2016). arXiv:1609.04836 arXiv preprint"},{"key":"12_CR18","unstructured":"Kingma, D.P., Ba, J.: Adam: a method for stochastic optimization (2014). arXiv:1412.6980 arXiv preprint"},{"key":"12_CR19","unstructured":"Krizhevsky, A., Hinton, G.: Learning multiple layers of features from tiny images. Tech. Rep. 0, University of Toronto, Toronto, Ontario (2009). https:\/\/www.cs.toronto.edu\/~kriz\/learning-features-2009-TR.pdf"},{"key":"12_CR20","unstructured":"Kwon, J., Kim, J., Park, H., Choi, I.K.: ASAM: adaptive sharpness-aware minimization for scale-invariant learning of deep neural networks. CoRR abs\/2102.11600 (2021). https:\/\/arxiv.org\/abs\/2102.11600"},{"key":"12_CR21","doi-asserted-by":"crossref","unstructured":"Li, T., Zhou, P., He, Z., Cheng, X., Huang, X.: Friendly sharpness-aware minimization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5631\u20135640 (2024)","DOI":"10.1109\/CVPR52733.2024.00538"},{"key":"12_CR22","unstructured":"Li, Y., Wei, C., Ma, T.: Towards explaining the regularization effect of initial large learning rate in training neural networks. Adv. Neural. Inf. Process. Syst. 32 (2019)"},{"key":"12_CR23","unstructured":"Liang, T., Poggio, T., Rakhlin, A., Stokes, J.: Fisher-RAO metric, geometry, and complexity of neural networks. In: The 22nd International Conference on Artificial Intelligence and Statistics, pp. 888\u2013896. PMLR (2019)"},{"key":"12_CR24","doi-asserted-by":"crossref","unstructured":"McAllester, D.A.: Pac-bayesian model averaging. In: Proceedings of the Twelfth Annual Conference on Computational Learning Theory, pp. 164\u2013170 (1999)","DOI":"10.1145\/307400.307435"},{"key":"12_CR25","unstructured":"Neyshabur, B., Bhojanapalli, S., Srebro, N.: A pac-bayesian approach to spectrallynormalized margin bounds for neural networks. arXiv preprint arXiv:1707.09564 (2017)"},{"issue":"4","key":"12_CR26","doi-asserted-by":"publisher","first-page":"838","DOI":"10.1137\/0330046","volume":"30","author":"BT Polyak","year":"1992","unstructured":"Polyak, B.T., Juditsky, A.B.: Acceleration of stochastic approximation by averaging. SIAM J. Control. Optim. 30(4), 838\u2013855 (1992). https:\/\/doi.org\/10.1137\/0330046","journal-title":"SIAM J. Control. Optim."},{"key":"12_CR27","doi-asserted-by":"publisher","unstructured":"Ratchatorn, T., Tanaka, M.: Adaptive adversarial cross-entropy loss for sharpness-aware minimization. In: 2024 IEEE International Conference on Image Processing (ICIP), pp. 479\u2013485. (2024). https:\/\/doi.org\/10.1109\/ICIP51287.2024.10647582","DOI":"10.1109\/ICIP51287.2024.10647582"},{"key":"12_CR28","doi-asserted-by":"publisher","unstructured":"Ratchatorn, T., Tanaka, M.: Improving sharpness-aware minimization using label smoothing and adaptive adversarial cross-entropy loss. IEEE Access 13, 100326\u2013100337 (2025). https:\/\/doi.org\/10.1109\/ACCESS.2025.3578265","DOI":"10.1109\/ACCESS.2025.3578265"},{"key":"12_CR29","doi-asserted-by":"crossref","unstructured":"Ratchatorn, T., Tanaka, M.: Flatface: improve face recognition by sharpness-aware minimization. In: 2026 Electronic Imaging (EI) (2026)","DOI":"10.2352\/EI.2026.38.6.ISS-278"},{"issue":"3","key":"12_CR30","doi-asserted-by":"publisher","first-page":"400","DOI":"10.1214\/aoms\/1177729586","volume":"22","author":"H Robbins","year":"1951","unstructured":"Robbins, H., Monro, S.: A stochastic approximation method. Ann. Math. Stat. 22(3), 400\u2013407 (1951)","journal-title":"Ann. Math. Stat."},{"key":"12_CR31","unstructured":"Ruppert, D.: Efficient estimations from a slowly convergent robbins-monro process. Tech. rep., Cornell University Operations Research and Industrial Engineering (1988)"},{"key":"12_CR32","unstructured":"Sato, I., Ishikawa, K., Liu, G., Tanaka, M.: Breaking inter-layer co-adaptation by classifier anonymization. In: International Conference on Machine Learning, pp. 5619\u20135627 (2019) PMLR"},{"key":"12_CR33","unstructured":"Sato, I., Ryota, Y., Tanaka, M., Inoue, N., Kawakami, R.: POF: post-training of feature extractor for improving generalization. In: International Conference on Machine Learning, pp. 19221\u201319230. PMLR (2022)"},{"key":"12_CR34","doi-asserted-by":"crossref","unstructured":"Szegedy, C., et al.: Going deeper with convolutions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1\u20139 (2015)","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"12_CR35","doi-asserted-by":"crossref","unstructured":"Szegedy, C., Vanhoucke, V., Ioffe, S., Shlens, J., Wojna, Z.: Rethinking the inception architecture for computer vision. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2818\u20132826 (2016)","DOI":"10.1109\/CVPR.2016.308"},{"key":"12_CR36","unstructured":"Xiao, H., Rasul, K., Vollgraf, R.: Fashion-MNIST: a novel image dataset for benchmarking machine learning algorithms (2017). arXiv:1708.07747 arXiv preprint"},{"key":"12_CR37","doi-asserted-by":"crossref","unstructured":"Zagoruyko, S., Komodakis, N.: Wide residual networks. arXiv preprint arXiv:1605.07146 (2016)","DOI":"10.5244\/C.30.87"},{"issue":"3","key":"12_CR38","doi-asserted-by":"publisher","first-page":"107","DOI":"10.1145\/3446776","volume":"64","author":"C Zhang","year":"2021","unstructured":"Zhang, C., Bengio, S., Hardt, M., Recht, B., Vinyals, O.: Understanding deep learning (still) requires rethinking generalization. Commun. ACM 64(3), 107\u2013115 (2021)","journal-title":"Commun. ACM"},{"key":"12_CR39","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Luo, R., Su, Q., Sun, X.: Ga-SAM: gradient-strength based adaptive sharpness-aware minimization for improved generalization. arXiv preprint arXiv:2210.06895 (2022)","DOI":"10.18653\/v1\/2022.emnlp-main.257"},{"key":"12_CR40","unstructured":"Zhao, Y., Zhang, H., Hu, X.: Ss-SAM : stochastic scheduled sharpness-aware minimization for efficiently training deep neural networks. arXiv preprint arXiv:2203.09962 (2022)"},{"key":"12_CR41","unstructured":"Zhuang, J., et al.: Surrogate gap minimization improves sharpness-aware training. arXiv preprint arXiv:2203.08065 (2022)"}],"container-title":["Lecture Notes in Computer Science","Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-31666-0_12","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,8,2]],"date-time":"2026-08-02T05:46:08Z","timestamp":1785649568000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-31666-0_12"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8,3]]},"ISBN":["9783032316653","9783032316660"],"references-count":41,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-31666-0_12","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,8,3]]},"assertion":[{"value":"3 August 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICPR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Pattern Recognition","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Lyon","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"France","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17 August 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 August 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icpr2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/icpr2026.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}