{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,26]],"date-time":"2025-03-26T16:25:28Z","timestamp":1743006328073,"version":"3.40.3"},"publisher-location":"Singapore","reference-count":34,"publisher":"Springer Nature Singapore","isbn-type":[{"type":"print","value":"9789819981250"},{"type":"electronic","value":"9789819981267"}],"license":[{"start":{"date-parts":[[2023,11,13]],"date-time":"2023-11-13T00:00:00Z","timestamp":1699833600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,11,13]],"date-time":"2023-11-13T00:00:00Z","timestamp":1699833600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024]]},"DOI":"10.1007\/978-981-99-8126-7_43","type":"book-chapter","created":{"date-parts":[[2023,11,24]],"date-time":"2023-11-24T08:01:53Z","timestamp":1700812913000},"page":"555-567","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Sharpness-Aware Minimization for\u00a0Out-of-Distribution Generalization"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5790-2473","authenticated-orcid":false,"given":"Dongqi","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1754-4878","authenticated-orcid":false,"given":"Zhu","family":"Teng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9940-8409","authenticated-orcid":false,"given":"Qirui","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-1512-9397","authenticated-orcid":false,"given":"Ziyin","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,11,13]]},"reference":[{"key":"43_CR1","unstructured":"Arjovsky, M., Bottou, L., Gulrajani, I., Lopez-Paz, D.: Invariant risk minimization. arXiv preprint arXiv:1907.02893 (2019)"},{"key":"43_CR2","unstructured":"Bisla, D., Wang, J., Choromanska, A.: Low-pass filtering SGD for recovering flat optima in the deep learning optimization landscape. In: International Conference on Artificial Intelligence and Statistics, pp. 8299\u20138339. PMLR (2022)"},{"key":"43_CR3","first-page":"22405","volume":"34","author":"J Cha","year":"2021","unstructured":"Cha, J., et al.: SWAD: domain generalization by seeking flat minima. Adv. Neural. Inf. Process. Syst. 34, 22405\u201322418 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"43_CR4","unstructured":"Chang, S., Zhang, Y., Yu, M., Jaakkola, T.: Invariant rationalization. In: International Conference on Machine Learning, pp. 1448\u20131458. PMLR (2020)"},{"key":"43_CR5","unstructured":"Chen, Y., et al.: Pareto invariant risk minimization: Towards mitigating the optimization dilemma in out-of-distribution generalization. In: The Eleventh International Conference on Learning Representations (2023)"},{"key":"43_CR6","unstructured":"Creager, E., Jacobsen, J.H., Zemel, R.: Environment inference for invariant learning. In: International Conference on Machine Learning, pp. 2189\u20132200. PMLR (2021)"},{"issue":"3","key":"43_CR7","doi-asserted-by":"publisher","first-page":"1378","DOI":"10.1214\/20-AOS2004","volume":"49","author":"JC Duchi","year":"2021","unstructured":"Duchi, J.C., Namkoong, H.: Learning models with uniform performance via distributionally robust optimization. Ann. Stat. 49(3), 1378\u20131406 (2021)","journal-title":"Ann. Stat."},{"key":"43_CR8","unstructured":"Foret, P., Kleiner, A., Mobahi, H., Neyshabur, B.: Sharpness-aware minimization for efficiently improving generalization. In: International Conference on Learning Representations (2021)"},{"key":"43_CR9","unstructured":"Gulrajani, I., Lopez-Paz, D.: In search of lost domain generalization. In: International Conference on Learning Representations (2021)"},{"key":"43_CR10","unstructured":"He, H., Huang, G., Yuan, Y.: Asymmetric valleys: beyond sharp and flat local minima. In: Advances in Neural Information Processing Systems, vol. 32 (2019)"},{"issue":"1","key":"43_CR11","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1162\/neco.1997.9.1.1","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter, S., Schmidhuber, J.: Flat minima. Neural Comput. 9(1), 1\u201342 (1997)","journal-title":"Neural Comput."},{"key":"43_CR12","unstructured":"Hu, Z., Hong, L.J.: Kullback-leibler divergence constrained distributionally robust optimization. Available at Optimization Online 1(2), 1695\u20131724 (2013)"},{"key":"43_CR13","doi-asserted-by":"crossref","unstructured":"Huang, Z., et al.: Robust generalization against photon-limited corruptions via worst-case sharpness minimization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16175\u201316185 (2023)","DOI":"10.1109\/CVPR52729.2023.01552"},{"key":"43_CR14","unstructured":"Izmailov, P., Podoprikhin, D., Garipov, T., Vetrov, D., Wilson, A.: Averaging weights leads to wider optima and better generalization. In: 34th Conference on Uncertainty in Artificial Intelligence 2018, UAI 2018, pp. 876\u2013885 (2018)"},{"key":"43_CR15","first-page":"16577","volume":"35","author":"J Kaddour","year":"2022","unstructured":"Kaddour, J., Liu, L., Silva, R., Kusner, M.J.: When do flat minima optimizers work? Adv. Neural. Inf. Process. Syst. 35, 16577\u201316595 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"43_CR16","unstructured":"Kamath, P., Tangella, A., Sutherland, D., Srebro, N.: Does invariant risk minimization capture invariance? In: International Conference on Artificial Intelligence and Statistics, pp. 4069\u20134077. PMLR (2021)"},{"key":"43_CR17","unstructured":"Keskar, N.S., Mudigere, D., Nocedal, J., Smelyanskiy, M., Tang, P.T.P.: On large-batch training for deep learning: Generalization gap and sharp minima. In: International Conference on Learning Representations (2017)"},{"key":"43_CR18","unstructured":"Kim, M., Li, D., Hu, S.X., Hospedales, T.: Fisher SAM: information geometry and sharpness aware minimisation. In: International Conference on Machine Learning, pp. 11148\u201311161. PMLR (2022)"},{"key":"43_CR19","unstructured":"Krueger, D., et al.: Out-of-distribution generalization via risk extrapolation (rex). In: International Conference on Machine Learning, pp. 5815\u20135826. PMLR (2021)"},{"key":"43_CR20","doi-asserted-by":"crossref","unstructured":"Lin, Y., Dong, H., Wang, H., Zhang, T.: Bayesian invariant risk minimization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16021\u201316030 (2022)","DOI":"10.1109\/CVPR52688.2022.01555"},{"key":"43_CR21","doi-asserted-by":"crossref","unstructured":"Liu, Z., Luo, P., Wang, X., Tang, X.: Deep learning face attributes in the wild. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 3730\u20133738 (2015)","DOI":"10.1109\/ICCV.2015.425"},{"key":"43_CR22","doi-asserted-by":"crossref","unstructured":"McAllester, D.A.: Pac-bayesian model averaging. In: Proceedings of the Twelfth Annual Conference on Computational Learning Theory, pp. 164\u2013170 (1999)","DOI":"10.1145\/307400.307435"},{"key":"43_CR23","unstructured":"Petzka, H., Kamp, M., Adilova, L., Sminchisescu, C., Boley, M.: Relative flatness and generalization. In: In: Advances in Neural Information Processing Systems (2021)"},{"key":"43_CR24","unstructured":"Rame, A., Dancette, C., Cord, M.: Fishr: invariant gradient variances for out-of-distribution generalization. In: International Conference on Machine Learning, pp. 18347\u201318377. PMLR (2022)"},{"key":"43_CR25","unstructured":"Rame, A., Kirchmeyer, M., Rahier, T., Rakotomamonjy, A., patrick gallinari, Cord, M.: Diverse weight averaging for out-of-distribution generalization. In: Advances in Neural Information Processing Systems (2022)"},{"key":"43_CR26","unstructured":"Rangwani, H., Aithal, S.K., Mishra, M., Jain, A., Radhakrishnan, V.B.: A closer look at smoothness in domain adversarial training. In: International Conference on Machine Learning, pp. 18378\u201318399. PMLR (2022)"},{"key":"43_CR27","unstructured":"Rosenfeld, E., Ravikumar, P., Risteski, A.: The risks of invariant risk minimization. arXiv preprint arXiv:2010.05761 (2020)"},{"key":"43_CR28","unstructured":"Sagawa, S., Koh, P.W., Hashimoto, T.B., Liang, P.: Distributionally robust neural networks for group shifts: On the importance of regularization for worst-case generalization. In: International Conference on Learning Representations (ICLR) (2020)"},{"key":"43_CR29","unstructured":"Sagawa, S., Raghunathan, A., Koh, P.W., Liang, P.: An investigation of why overparameterization exacerbates spurious correlations. In: International Conference on Machine Learning, pp. 8346\u20138356. PMLR (2020)"},{"key":"43_CR30","unstructured":"Vapnik, V.: Statistical Learning Theory. Wiley, New York (1998)"},{"key":"43_CR31","doi-asserted-by":"crossref","unstructured":"Ye, N., et al.: OoD-bench: quantifying and understanding two dimensions of out-of-distribution generalization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7947\u20137958 (2022)","DOI":"10.1109\/CVPR52688.2022.00779"},{"key":"43_CR32","unstructured":"Yong, L., Zhu, S., Tan, L., Cui, P.: Zin: When and how to learn invariance without environment partition? In: Advances in Neural Information Processing Systems (2022)"},{"key":"43_CR33","unstructured":"Zhang, Y., Sharma, P., Ram, P., Hong, M., Varshney, K.R., Liu, S.: What is missing in IRM training and evaluation? challenges and solutions. In: The Eleventh International Conference on Learning Representations (2023)"},{"key":"43_CR34","unstructured":"Zhou, X., Lin, Y., Zhang, W., Zhang, T.: Sparse invariant risk minimization. In: International Conference on Machine Learning, pp. 27222\u201327244. PMLR (2022)"}],"container-title":["Communications in Computer and Information Science","Neural Information Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-99-8126-7_43","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,7]],"date-time":"2024-03-07T11:39:22Z","timestamp":1709811562000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-99-8126-7_43"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,11,13]]},"ISBN":["9789819981250","9789819981267"],"references-count":34,"URL":"https:\/\/doi.org\/10.1007\/978-981-99-8126-7_43","relation":{},"ISSN":["1865-0929","1865-0937"],"issn-type":[{"type":"print","value":"1865-0929"},{"type":"electronic","value":"1865-0937"}],"subject":[],"published":{"date-parts":[[2023,11,13]]},"assertion":[{"value":"13 November 2023","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICONIP","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Neural Information Processing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Changsha","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2023","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"20 November 2023","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 November 2023","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"30","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"iconip2023","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/iconip2023.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Single-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"EasyChair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"1274","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"650","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"51% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"4.14","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"2.46","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}