{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,10]],"date-time":"2026-02-10T16:09:06Z","timestamp":1770739746394,"version":"3.49.0"},"reference-count":73,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"3","license":[{"start":{"date-parts":[[2026,3,1]],"date-time":"2026-03-01T00:00:00Z","timestamp":1772323200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,3,1]],"date-time":"2026-03-01T00:00:00Z","timestamp":1772323200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,3,1]],"date-time":"2026-03-01T00:00:00Z","timestamp":1772323200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100018537","name":"National Science and Technology Major Project","doi-asserted-by":"publisher","award":["2022ZD0115101"],"award-info":[{"award-number":["2022ZD0115101"]}],"id":[{"id":"10.13039\/501100018537","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Research Center for Industries of the Future"},{"DOI":"10.13039\/100018928","name":"Westlake University","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100018928","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100032572","name":"Westlake Education Foundation","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100032572","id-type":"DOI","asserted-by":"publisher"}]},{"name":"National Key Research and Development Project of China","award":["2021ZD0110505"],"award-info":[{"award-number":["2021ZD0110505"]}]},{"name":"Zhejiang Provincial Key Research and Development Project","award":["2023C01043"],"award-info":[{"award-number":["2023C01043"]}]},{"name":"Academy of Social Governance Zhejiang University"},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62441617"],"award-info":[{"award-number":["62441617"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Zhejiang Provincial Natural Science Foundation of China","award":["LD25F020001"],"award-info":[{"award-number":["LD25F020001"]}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","award":["226-2025-00057"],"award-info":[{"award-number":["226-2025-00057"]}],"id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Pattern Anal. Mach. Intell."],"published-print":{"date-parts":[[2026,3]]},"DOI":"10.1109\/tpami.2025.3624314","type":"journal-article","created":{"date-parts":[[2025,10,22]],"date-time":"2025-10-22T17:24:24Z","timestamp":1761153864000},"page":"2186-2199","source":"Crossref","is-referenced-by-count":0,"title":["Improving Model Fusion by Training-Time Neuron Alignment With Fixed Neuron Anchors"],"prefix":"10.1109","volume":"48","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0831-3549","authenticated-orcid":false,"given":"Zexi","family":"Li","sequence":"first","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6892-8097","authenticated-orcid":false,"given":"Zhiqi","family":"Li","sequence":"additional","affiliation":[{"name":"Georgia Institute of Technology, Atlanta, GA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jie","family":"Lin","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tao","family":"Shen","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6142-9914","authenticated-orcid":false,"given":"Jun","family":"Xiao","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3075-2161","authenticated-orcid":false,"given":"Yike","family":"Guo","sequence":"additional","affiliation":[{"name":"Hong Kong University of Science and Technology, Hong Kong SAR, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3246-6935","authenticated-orcid":false,"given":"Tao","family":"Lin","sequence":"additional","affiliation":[{"name":"Westlake University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0885-6869","authenticated-orcid":false,"given":"Chao","family":"Wu","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"issue":"8","key":"ref1","first-page":"9","article-title":"Language models are unsupervised multitask learners","volume":"1","author":"Radford","year":"2019","journal-title":"OpenAI blog"},{"key":"ref2","first-page":"1877","article-title":"Language models are few-shot learners","volume":"33","author":"Brown","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"ref3","article-title":"Llama 2: Open foundation and fine-tuned chat models","author":"Touvron","year":"2023"},{"key":"ref4","article-title":"GPT-4 technical report","year":"2023"},{"key":"ref5","article-title":"Hierarchical text-conditional image generation with clip latents","author":"Ramesh","year":"2022"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"ref7","article-title":"Creating video from text","year":"2024"},{"key":"ref8","article-title":"Fusionbench: A comprehensive benchmark of deep model fusion","author":"Tang","year":"2024"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/tnnls.2025.3628666"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-industry.36"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.46"},{"key":"ref12","first-page":"23965","article-title":"Model soups: Averaging weights of multiple fine-tuned models improves accuracy without increasing inference time","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Wortsman","year":"2022"},{"key":"ref13","first-page":"1273","article-title":"Communication-efficient learning of deep networks from decentralized data","volume-title":"Proc. Artif. Intell. Statist.","author":"McMahan","year":"2017"},{"key":"ref14","article-title":"Federated learning with matched averaging","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Wang","year":"2020"},{"key":"ref15","first-page":"788","article-title":"Revisiting weighted aggregation in federated learning with neural networks","volume-title":"Proc. 40th Int. Conf. Mach. Learn.","volume":"202","author":"Li","year":"2023"},{"key":"ref16","first-page":"3259","article-title":"Linear mode connectivity and the lottery ticket hypothesis","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Frankle","year":"2020"},{"key":"ref17","first-page":"1309","article-title":"Essentially no barriers in neural network energy landscape","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Draxler","year":"2018"},{"key":"ref18","article-title":"The role of permutation invariance in linear mode connectivity of neural networks","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Entezari","year":"2022"},{"key":"ref19","article-title":"Git re-basin: Merging models modulo permutation symmetries","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Ainsworth","year":"2022"},{"key":"ref20","first-page":"21766","article-title":"Ties-merging: Resolving interference when merging models","volume":"36","author":"Yadav","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"ref21","article-title":"An image is worth 16 \u00d7 16 words: Transformers for image recognition at scale","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Dosovitskiy","year":"2021"},{"key":"ref22","article-title":"Improving language understanding by generative pre-training","volume-title":"OpenAI","author":"Radford","year":"2018"},{"key":"ref23","first-page":"22045","article-title":"Model fusion via optimal transport","volume":"33","author":"Singh","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"ref24","article-title":"Transformer fusion with optimal transport","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Imfeld","year":"2024"},{"key":"ref25","article-title":"KAN: Kolmogorov-Arnold networks","volume-title":"Proc. 13th Int. Conf. Learn. Representations","author":"Liu","year":"2025"},{"key":"ref26","article-title":"Mamba:Linear-time sequence modeling with selective state spaces","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Gu","year":"2024"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.632"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00490"},{"key":"ref29","article-title":"Federated learning based on dynamic regularization","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Acar","year":"2020"},{"key":"ref30","first-page":"8794","article-title":"Loss surfaces, mode connectivity, and fast ensembling of dnns","volume":"31","author":"Garipov","year":"2018","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"ref31","first-page":"769","article-title":"Loss surface simplexes for mode connecting volumes and fast ensembling","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Benton","year":"2021"},{"key":"ref32","article-title":"Editing models with task arithmetic","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Ilharco","year":"2023"},{"key":"ref33","first-page":"51838","article-title":"Task arithmetic in the tangent space: Improved editing of pre-trained models","volume":"36","author":"Ortiz-Jimenez","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"ref34","first-page":"341","article-title":"What can linear interpolation of neural network loss landscapes tell us?","volume-title":"Proc. Int. Conf. Mach. Learn.","volume":"22","author":"Vlaar","year":"2022"},{"key":"ref35","article-title":"Layerwise linear mode connectivity","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Adilova","year":"2024"},{"key":"ref36","article-title":"Bridging mode connectivity in loss landscapes and adversarial robustness","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Zhao","year":"2020"},{"key":"ref37","first-page":"22965","article-title":"Mechanistic mode connectivity","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Lubana","year":"2023"},{"key":"ref38","first-page":"9722","article-title":"Geometry of the loss landscape in overparameterized neural networks: Symmetries and invariances","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Simsek","year":"2021"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1016\/B978-0-444-88400-8.50019-4"},{"key":"ref40","first-page":"7252","article-title":"Bayesian nonparametric federated learning of neural networks","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Yurochkin","year":"2019"},{"key":"ref41","article-title":"Federated learning with matched averaging","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Wang","year":"2020"},{"key":"ref42","first-page":"15300","article-title":"Optimizing mode connectivity via neuron alignment","volume":"33","author":"Tatro","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"ref43","first-page":"13857","article-title":"Deep neural network fusion via graph matching with applications to model ensemble and federated learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Liu","year":"2022"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01938"},{"key":"ref45","first-page":"39879","article-title":"Feddisco: Federated learning with discrepancy-aware collaboration","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Ye","year":"2023"},{"key":"ref46","first-page":"2351","article-title":"Ensemble distillation for robust model fusion in federated learning","volume":"33","author":"Lin","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00984"},{"key":"ref48","article-title":"Linear mode connectivity in sparse neural networks","volume":"37","author":"McDermott","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"ref49","article-title":"Sparse weight averaging with multiple parti-cles for iterative magnitude pruning","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Choi","year":"2024"},{"key":"ref50","article-title":"Deep gradient compression: Reducing the communication bandwidth for distributed training","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Lin","year":"2018"},{"key":"ref51","first-page":"6318","article-title":"Powersgd: Practical low-rank gradient compression for distributed optimization","volume":"32","author":"Vogels","year":"2019","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"ref52","article-title":"MNIST handwritten digit database","author":"LeCun","year":"2010"},{"key":"ref53","volume-title":"Learning Multiple Layers of Features From Tiny Images.","author":"Krizhevsky","year":"2009"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref55","article-title":"Continual learning with hypernetworks","volume-title":"Proc. Int. Conf. Learn. Representations","author":"von Oswald","year":"2019"},{"key":"ref56","article-title":"Omnigrok: Grokking beyond algorithmic data","volume-title":"Proc. 11th Int. Conf. Learn. Representations","author":"Liu","year":"2022"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-24797-2"},{"key":"ref58","first-page":"142","article-title":"Learning word vectors for sentiment analysis","volume-title":"Proc. 49th Annu. meeting Assoc. Comput. Linguistics: Hum. Lang. Technol.","author":"Maas","year":"2011"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"ref60","article-title":"Tiny ImageNet dataset","year":"2023"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1016\/S0167-7152(96)00140-X"},{"key":"ref62","article-title":"Pointer sentinel mixture models","volume-title":"Proc. 5th Int. Conf. Learn. Representations (ICLR)","author":"Merity","year":"2017"},{"key":"ref63","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"Proc. 38th Int. Conf. Mach. Learn.","volume":"139","author":"Radford","year":"2021"},{"key":"ref64","first-page":"1135","article-title":"Learning both weights and connections for efficient neural network","volume":"28","author":"Han","year":"2015","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"ref65","article-title":"Deep compression: Compressing deep neural networks with pruning, trained quantization and Huffman coding","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Han","year":"2016"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/w18-5446"},{"key":"ref67","first-page":"7611","article-title":"Tackling the objective inconsistency problem in heterogeneous federated optimization","volume":"33","author":"Wang","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"ref68","first-page":"5132","article-title":"Scaffold: Stochastic controlled averaging for federated learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Karimireddy","year":"2020"},{"key":"ref69","article-title":"Fashion-mnist: A novel image dataset for benchmarking machine learning algorithms","author":"Xiao","year":"2017"},{"key":"ref70","article-title":"On bridging generic and personalized federated learning for image classification","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Chen","year":"2022"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/5.726791"},{"key":"ref72","first-page":"429","article-title":"Federated optimization in heterogeneous networks","volume-title":"Proc. Mach. Learn. Syst.","volume":"2","author":"Li","year":"2020"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i6.25891"}],"container-title":["IEEE Transactions on Pattern Analysis and Machine Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/34\/11372200\/11214255.pdf?arnumber=11214255","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,2,9]],"date-time":"2026-02-09T21:06:08Z","timestamp":1770671168000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11214255\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3]]},"references-count":73,"journal-issue":{"issue":"3"},"URL":"https:\/\/doi.org\/10.1109\/tpami.2025.3624314","relation":{},"ISSN":["0162-8828","2160-9292","1939-3539"],"issn-type":[{"value":"0162-8828","type":"print"},{"value":"2160-9292","type":"electronic"},{"value":"1939-3539","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,3]]}}}