{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,23]],"date-time":"2026-06-23T08:47:04Z","timestamp":1782204424346,"version":"3.54.5"},"reference-count":46,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012165","name":"Key Technologies Research and Development Program","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100012165","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Pattern Recognition"],"published-print":{"date-parts":[[2026,12]]},"DOI":"10.1016\/j.patcog.2026.114262","type":"journal-article","created":{"date-parts":[[2026,6,19]],"date-time":"2026-06-19T23:18:34Z","timestamp":1781911114000},"page":"114262","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PC","title":["Geometric insights into the relationship between weight landscape and generalization"],"prefix":"10.1016","volume":"180","author":[{"given":"Zhixing","family":"Lu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bo","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6515-134X","authenticated-orcid":false,"given":"Yuanyuan","family":"Sun","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuanyu","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-0432-692X","authenticated-orcid":false,"given":"Paerhati","family":"Tulajiang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0872-7688","authenticated-orcid":false,"given":"Hongfei","family":"Lin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.patcog.2026.114262_b1","series-title":"Proceedings of the 41st International Conference on Machine Learning","first-page":"20983","article-title":"Understanding the learning dynamics of alignment with human feedback","volume":"vol. 235","author":"Im","year":"2024"},{"key":"10.1016\/j.patcog.2026.114262_b2","doi-asserted-by":"crossref","unstructured":"X. Zhang, J. Li, W. Chu, R. Xu, Y. Yang, S. Guan, J. Xu, L. Jing, P. Cui, et al., On the out-of-distribution generalization of large multimodal models, in: Proceedings of the Computer Vision and Pattern Recognition Conference, 2025, pp. 10315\u201310326.","DOI":"10.1109\/CVPR52734.2025.00965"},{"issue":"4","key":"10.1016\/j.patcog.2026.114262_b3","doi-asserted-by":"crossref","first-page":"3067","DOI":"10.1109\/TPAMI.2025.3528193","article-title":"Adaptive biased stochastic optimization","volume":"47","author":"Yang","year":"2025","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.patcog.2026.114262_b4","series-title":"The llama 3 herd of models","author":"Dubey","year":"2024"},{"key":"10.1016\/j.patcog.2026.114262_b5","series-title":"Gpt-4 technical report","author":"Achiam","year":"2023"},{"issue":"240","key":"10.1016\/j.patcog.2026.114262_b6","first-page":"1","article-title":"Palm: Scaling language modeling with pathways","volume":"24","author":"Chowdhery","year":"2023","journal-title":"J. Mach. Learn. Res."},{"key":"10.1016\/j.patcog.2026.114262_b7","article-title":"The geometry of neural nets\u2019 parameter spaces under reparametrization","volume":"36","author":"Kristiadi","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.114262_b8","unstructured":"M. Mustaq, X.M. Tricoche, Globscope: Toward a Global View of the Loss Landscape, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2026, pp. 5263\u20135272."},{"key":"10.1016\/j.patcog.2026.114262_b9","series-title":"International Conference on Machine Learning","first-page":"22325","article-title":"What can linear interpolation of neural network loss landscapes tell us?","author":"Vlaar","year":"2022"},{"key":"10.1016\/j.patcog.2026.114262_b10","doi-asserted-by":"crossref","DOI":"10.1016\/j.neunet.2025.107507","article-title":"On neural architecture search and hyperparameter optimization: A max-flow based approach","volume":"188","author":"Xue","year":"2025","journal-title":"Neural Netw."},{"key":"10.1016\/j.patcog.2026.114262_b11","doi-asserted-by":"crossref","unstructured":"J. Devlin, M.-W. Chang, K. Lee, K. Toutanova, Bert: Pre-training of deep bidirectional transformers for language understanding, in: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers), 2019, pp. 4171\u20134186.","DOI":"10.18653\/v1\/N19-1423"},{"issue":"8","key":"10.1016\/j.patcog.2026.114262_b12","first-page":"9","article-title":"Language models are unsupervised multitask learners","volume":"1","author":"Radford","year":"2019","journal-title":"OpenAI Blog"},{"key":"10.1016\/j.patcog.2026.114262_b13","series-title":"Qwen technical report","author":"Bai","year":"2023"},{"key":"10.1016\/j.patcog.2026.114262_b14","series-title":"Glm-130b: An open bilingual pre-trained model","author":"Zeng","year":"2022"},{"key":"10.1016\/j.patcog.2026.114262_b15","doi-asserted-by":"crossref","first-page":"46595","DOI":"10.52202\/075280-2020","article-title":"Judging llm-as-a-judge with mt-bench and chatbot arena","volume":"36","author":"Zheng","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.114262_b16","series-title":"Llama3-8B-Chinese-Chat (revision 6622a23)","author":"Wang","year":"2024"},{"key":"10.1016\/j.patcog.2026.114262_b17","series-title":"Proceedings of the Fourteenth International Conference on Artificial Intelligence and Statistics","first-page":"315","article-title":"Deep sparse rectifier neural networks","author":"Glorot","year":"2011"},{"key":"10.1016\/j.patcog.2026.114262_b18","unstructured":"X. Glorot, Y. Bengio, Understanding the difficulty of training deep feedforward neural networks, in: Proceedings of the Thirteenth International Conference on Artificial Intelligence and Statistics, 2010, pp. 249\u2013256."},{"key":"10.1016\/j.patcog.2026.114262_b19","doi-asserted-by":"crossref","unstructured":"K. He, X. Zhang, S. Ren, J. Sun, Delving deep into rectifiers: Surpassing human-level performance on imagenet classification, in: Proceedings of the IEEE International Conference on Computer Vision, 2015, pp. 1026\u20131034.","DOI":"10.1109\/ICCV.2015.123"},{"key":"10.1016\/j.patcog.2026.114262_b20","series-title":"International Conference on Machine Learning","first-page":"2798","article-title":"Geometry of neural network loss surfaces via random matrix theory","author":"Pennington","year":"2017"},{"key":"10.1016\/j.patcog.2026.114262_b21","series-title":"Topological, Algebraic and Geometric Learning Workshops 2023","first-page":"437","article-title":"Learning to see topological properties in 4d using convolutional neural networks","author":"Hannouch","year":"2023"},{"key":"10.1016\/j.patcog.2026.114262_b22","article-title":"Visualizing the loss landscape of neural nets","volume":"31","author":"Li","year":"2018","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.114262_b23","series-title":"On large-batch training for deep learning: gap and sharp minima","author":"Keskar","year":"2016"},{"key":"10.1016\/j.patcog.2026.114262_b24","unstructured":"D.P. Kingma, J. Ba, Adam: A Method for Stochastic Optimization, in: 3rd International Conference on Learning Representations, ICLR 2015, San Diego, CA, USA, May 7-9, 2015, Conference Track Proceedings, 2015."},{"key":"10.1016\/j.patcog.2026.114262_b25","doi-asserted-by":"crossref","first-page":"400","DOI":"10.1214\/aoms\/1177729586","article-title":"A stochastic approximation method","author":"Robbins","year":"1951","journal-title":"Ann. Math. Stat."},{"key":"10.1016\/j.patcog.2026.114262_b26","article-title":"How does adaptive optimization impact local neural network geometry?","volume":"36","author":"Jiang","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.114262_b27","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.111105","article-title":"Approximate geometric structure transfer for cross-domain image classification","volume":"159","author":"Wong","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.114262_b28","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2023.110147","article-title":"From patch, sample to domain: Capture geometric structures for few-shot learning","volume":"148","author":"Li","year":"2024","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.114262_b29","article-title":"A simple weight decay can improve generalization","volume":"4","author":"Krogh","year":"1991","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.114262_b30","first-page":"4790","article-title":"On the training dynamics of deep networks with L_2 regularization","volume":"33","author":"Lewkowycz","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.114262_b31","series-title":"International Conference on Machine Learning","first-page":"152","article-title":"SAM operates far from home: eigenvalue regularization as a dynamical phenomenon","author":"Agarwala","year":"2023"},{"key":"10.1016\/j.patcog.2026.114262_b32","doi-asserted-by":"crossref","first-page":"6763","DOI":"10.52202\/068431-0490","article-title":"Feature learning in L_2-regularized DNNs: Attraction\/repulsion and sparsity","volume":"35","author":"Jacot","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.114262_b33","doi-asserted-by":"crossref","first-page":"9233","DOI":"10.52202\/068431-0671","article-title":"Fast mixing of stochastic gradient descent with normalization and weight decay","volume":"35","author":"Li","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.114262_b34","series-title":"Proceedings of the Thirty-Third International Joint Conference on Artificial Intelligence","first-page":"4687","article-title":"The orthogonality of weight vectors: The key characteristics of normalization and residual connections","author":"Lu","year":"2024"},{"key":"10.1016\/j.patcog.2026.114262_b35","doi-asserted-by":"crossref","unstructured":"T. Han, D. Gokay, J. Heyward, C. Zhang, D. Zoran, V. Patraucean, J. Carreira, D. Damen, A. Zisserman, Learning from streaming video with orthogonal gradients, in: Proceedings of the Computer Vision and Pattern Recognition Conference, 2025, pp. 13651\u201313660.","DOI":"10.1109\/CVPR52734.2025.01274"},{"issue":"4","key":"10.1016\/j.patcog.2026.114262_b36","doi-asserted-by":"crossref","first-page":"1352","DOI":"10.1109\/TPAMI.2019.2948352","article-title":"Orthogonal deep neural networks","volume":"43","author":"Li","year":"2019","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.patcog.2026.114262_b37","series-title":"Matrix Analysis","author":"Horn","year":"2012"},{"key":"10.1016\/j.patcog.2026.114262_b38","series-title":"Inequalities","author":"Hardy","year":"1952"},{"key":"10.1016\/j.patcog.2026.114262_b39","series-title":"Mathematics for Machine Learning","author":"Deisenroth","year":"2020"},{"issue":"6690","key":"10.1016\/j.patcog.2026.114262_b40","doi-asserted-by":"crossref","first-page":"1461","DOI":"10.1126\/science.adi5639","article-title":"Mechanism for feature learning in neural networks and backpropagation-free machine learning models","volume":"383","author":"Radhakrishnan","year":"2024","journal-title":"Science"},{"issue":"3","key":"10.1016\/j.patcog.2026.114262_b41","doi-asserted-by":"crossref","first-page":"492","DOI":"10.1002\/nla.1839","article-title":"Gram-Schmidt orthogonalization: 100 years and more","volume":"20","author":"Leon","year":"2013","journal-title":"Numer. Linear Algebra Appl."},{"key":"10.1016\/j.patcog.2026.114262_b42","unstructured":"I. Loshchilov, F. Hutter, Decoupled Weight Decay Regularization, in: International Conference on Learning Representations, 2019."},{"key":"10.1016\/j.patcog.2026.114262_b43","unstructured":"T. Tieleman, G. Hinton, Divide the Gradient by a Running Average of Its Recent Magnitude. Coursera: Neural Networks for Machine Learning, Technical Report, 2017."},{"key":"10.1016\/j.patcog.2026.114262_b44","doi-asserted-by":"crossref","unstructured":"K. He, X. Zhang, S. Ren, J. Sun, Deep residual learning for image recognition, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2016, pp. 770\u2013778.","DOI":"10.1109\/CVPR.2016.90"},{"key":"10.1016\/j.patcog.2026.114262_b45","series-title":"Advances in Neural Information Processing Systems","article-title":"Attention is all you need","volume":"Vol. 30","author":"Vaswani","year":"2017"},{"key":"10.1016\/j.patcog.2026.114262_b46","unstructured":"E.J. Hu, Y. Shen, P. Wallis, Z. Allen-Zhu, Y. Li, S. Wang, L. Wang, W. Chen, LoRA: Low-Rank Adaptation of Large Language Models, in: International Conference on Learning Representations, 2022."}],"container-title":["Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326012276?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326012276?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,23]],"date-time":"2026-06-23T07:51:36Z","timestamp":1782201096000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0031320326012276"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,12]]},"references-count":46,"alternative-id":["S0031320326012276"],"URL":"https:\/\/doi.org\/10.1016\/j.patcog.2026.114262","relation":{},"ISSN":["0031-3203"],"issn-type":[{"value":"0031-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,12]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Geometric insights into the relationship between weight landscape and generalization","name":"articletitle","label":"Article Title"},{"value":"Pattern Recognition","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.patcog.2026.114262","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"114262"}}