{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,5]],"date-time":"2026-05-05T15:28:41Z","timestamp":1777994921793,"version":"3.51.4"},"publisher-location":"Cham","reference-count":52,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031736490","type":"print"},{"value":"9783031736506","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,11,21]],"date-time":"2024-11-21T00:00:00Z","timestamp":1732147200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,21]],"date-time":"2024-11-21T00:00:00Z","timestamp":1732147200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-73650-6_20","type":"book-chapter","created":{"date-parts":[[2024,11,20]],"date-time":"2024-11-20T18:18:20Z","timestamp":1732126700000},"page":"342-359","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Learning Scalable Model Soup on\u00a0a\u00a0Single GPU: An Efficient Subspace Training Strategy"],"prefix":"10.1007","author":[{"given":"Tao","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Weisen","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fanghui","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaolin","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"James T.","family":"Kwok","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,11,21]]},"reference":[{"key":"20_CR1","unstructured":"Cai, R., Zhang, Z., Wang, Z.: Robust weight signatures: gaining robustness as easy as patching weights? In: International Conference on Machine Learning (ICML) (2023)"},{"key":"20_CR2","unstructured":"Camuto, A., Deligiannidis, G., Erdogdu, M.A., Gurbuzbalaban, M., Simsekli, U., Zhu, L.: Fractal structure and generalization properties of stochastic optimization algorithms. In: Advanced in Neural Information Processing Systems (NeurIPS) (2021)"},{"key":"20_CR3","doi-asserted-by":"crossref","unstructured":"Chen, M., Jiang, M., Dou, Q., Wang, Z., Li, X.: FedSoup: improving generalization and personalization in federated learning via selective model interpolation. In: International Conference on Medical Image Computing and Computer-Assisted Intervention (MICCA) (2023)","DOI":"10.1007\/978-3-031-43895-0_30"},{"key":"20_CR4","doi-asserted-by":"crossref","unstructured":"Chronopoulou, A., Peters, M.E., Fraser, A., Dodge, J.: AdapterSoup: weight averaging to improve generalization of pretrained language models. arXiv preprint arXiv:2302.07027 (2023)","DOI":"10.18653\/v1\/2023.findings-eacl.153"},{"key":"20_CR5","doi-asserted-by":"crossref","unstructured":"Croce, F., Rebuffi, S.A., Shelhamer, E., Gowal, S.: Seasoning model soups for robustness to adversarial and natural distribution shifts. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2023)","DOI":"10.1109\/CVPR52729.2023.01185"},{"key":"20_CR6","doi-asserted-by":"crossref","unstructured":"Ghadimi, S., Lan, G.: Stochastic first-and zeroth-order methods for nonconvex stochastic programming. SIAM J. Optim. (2013)","DOI":"10.1137\/120880811"},{"key":"20_CR7","doi-asserted-by":"crossref","unstructured":"Girshick, R., Donahue, J., Darrell, T., Malik, J.: Rich feature hierarchies for accurate object detection and semantic segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2014)","DOI":"10.1109\/CVPR.2014.81"},{"key":"20_CR8","unstructured":"Gressmann, F., Eaton-Rosen, Z., Luschi, C.: Improving neural network training in low dimensional random bases. In: Advances in Neural Information Processing Systems (NeurIPS) (2020)"},{"key":"20_CR9","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"20_CR10","unstructured":"Huang, B.: Adversarial learned soups: neural network averaging for joint clean and robust performance. Ph.D. thesis, Massachusetts Institute of Technology (2023)"},{"key":"20_CR11","doi-asserted-by":"crossref","unstructured":"Hunter, J.S.: The exponentially weighted moving average. J. Qual. Technol. (1986)","DOI":"10.1080\/00224065.1986.11979014"},{"key":"20_CR12","unstructured":"Ilharco, G., Ribeiro, M.T., Wortsman, M., Schmidt, L., Hajishirzi, H., Farhadi, A.: Editing models with task arithmetic. In: International Conference on Learning Representations (ICLR) (2023)"},{"key":"20_CR13","unstructured":"Ilharco, G., et al.: Patching open-vocabulary models by interpolating weights. In: Advances in Neural Information Processing Systems (NeurIPS) (2022)"},{"key":"20_CR14","unstructured":"Izmailov, P., Wilson, A., Podoprikhin, D., Vetrov, D., Garipov, T.: Averaging weights leads to wider optima and better generalization. In: Proceedings of Conference on Uncertainty in Artificial Intelligence (UAI) (2018)"},{"key":"20_CR15","unstructured":"Jiang, W., Kwok, J., Zhang, Y.: Subspace learning for effective meta-learning. In: International Conference on Machine Learning (ICML) (2022)"},{"key":"20_CR16","unstructured":"Jiang, W., Yang, H., Zhang, Y., Kwok, J.: An adaptive policy to employ sharpness-aware minimization. In: International Conference on Learning Representations (ICLR) (2023)"},{"key":"20_CR17","unstructured":"Kaddour, J.: Stop wasting my time! Saving days of imagenet and bert training with latest weight averaging. arXiv preprint arXiv:2209.14981 (2022)"},{"key":"20_CR18","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"491","DOI":"10.1007\/978-3-030-58558-7_29","volume-title":"Computer Vision \u2013 ECCV 2020","author":"A Kolesnikov","year":"2020","unstructured":"Kolesnikov, A., et al.: Big transfer (BiT): general visual representation learning. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12350, pp. 491\u2013507. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58558-7_29"},{"key":"20_CR19","doi-asserted-by":"crossref","unstructured":"Kornblith, S., Shlens, J., Le, Q.V.: Do better imagenet models transfer better? In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2019)","DOI":"10.1109\/CVPR.2019.00277"},{"key":"20_CR20","unstructured":"Lei, Y.: Stability and generalization of stochastic optimization with nonconvex and nonsmooth problems. In: The Thirty Sixth Annual Conference on Learning Theory (2023)"},{"key":"20_CR21","unstructured":"Li, C., Farkhoor, H., Liu, R., Yosinski, J.: Measuring the intrinsic dimension of objective landscapes. In: International Conference on Learning Representations (ICLR) (2018)"},{"key":"20_CR22","unstructured":"Li, T., Huang, Z., Tao, Q., Wu, Y., Huang, X.: Trainable weight averaging: efficient training by optimizing historical solutions. In: International Conference on Learning Representations (ICLR) (2022)"},{"key":"20_CR23","doi-asserted-by":"crossref","unstructured":"Li, T., Tan, L., Huang, Z., Tao, Q., Liu, Y., Huang, X.: Low dimensional trajectory hypothesis is true: dnns can be trained in tiny subspaces. IEEE Trans. Pattern Anal. Mach. Intell. (TPAMI) (2022)","DOI":"10.1109\/TPAMI.2022.3178101"},{"key":"20_CR24","unstructured":"Li, T., et al.: Revisiting random weight perturbation for efficiently improving generalization. Trans. Mach. Learn. Res. (TMLR) (2024)"},{"key":"20_CR25","doi-asserted-by":"crossref","unstructured":"Li, T., Wu, Y., Chen, S., Fang, K., Huang, X.: Subspace adversarial training. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2022)","DOI":"10.1109\/CVPR52688.2022.01305"},{"key":"20_CR26","doi-asserted-by":"crossref","unstructured":"Li, T., Zhou, P., He, Z., Cheng, X., Huang, X.: Friendly sharpness-aware minimization. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2024)","DOI":"10.1109\/CVPR52733.2024.00538"},{"key":"20_CR27","doi-asserted-by":"crossref","unstructured":"Liu, T.Y., Soatto, S.: Tangent model composition for ensembling and continual fine-tuning. In: IEEE\/CVF International Conference on Computer Vision (CVPR) (2023)","DOI":"10.1109\/ICCV51070.2023.01712"},{"key":"20_CR28","unstructured":"Melis, G.: Two-tailed averaging: anytime adaptive once-in-a-while optimal iterate averaging for stochastic optimization. arXiv preprint arXiv:2209.12581 (2022)"},{"key":"20_CR29","doi-asserted-by":"crossref","unstructured":"Nesterov, Y.: Efficiency of coordinate descent methods on huge-scale optimization problems. SIAM J. Optim. (2012)","DOI":"10.1137\/100802001"},{"key":"20_CR30","unstructured":"Ortiz-Jimenez, G., Favero, A., Frossard, P.: Task arithmetic in the tangent space: improved editing of pre-trained models. In: Advanced in Neural Information Processing Systems (NeurIPS) (2023)"},{"key":"20_CR31","unstructured":"Radford, A., et\u00a0al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning (ICML) (2021)"},{"key":"20_CR32","unstructured":"Rame, A., Kirchmeyer, M., Rahier, T., Rakotomamonjy, A., Gallinari, P., Cord, M.: Diverse weight averaging for out-of-distribution generalization. In: Advances in Neural Information Processing Systems (NeurIPS) (2022)"},{"key":"20_CR33","unstructured":"Ram\u00e9, A., et al.: Warm: on the benefits of weight averaged reward models. arXiv preprint arXiv:2401.12187 (2024)"},{"key":"20_CR34","unstructured":"Rebuffi, S.A., Croce, F., Gowal, S.: Revisiting adapters with adversarial training. In: International Conference on Learning Representations (ICLR) (2023)"},{"key":"20_CR35","doi-asserted-by":"crossref","unstructured":"Reddi, S.J., Hefny, A., Sra, S., Poczos, B., Smola, A.: Stochastic variance reduction for nonconvex optimization. In: International Conference on Machine Learning (ICML) (2016)","DOI":"10.1109\/ALLERTON.2016.7852377"},{"key":"20_CR36","unstructured":"Richt\u00e1rik, P., Tak\u00e1\u010d, M.: Distributed coordinate descent method for learning with big data. J. Mach. Learn. Res. (2016)"},{"key":"20_CR37","doi-asserted-by":"crossref","unstructured":"Richt\u00e1rik, P., Tak\u00e1\u010d, M.: Parallel coordinate descent methods for big data optimization. Math. Program. (2016)","DOI":"10.1007\/s10107-015-0901-6"},{"key":"20_CR38","doi-asserted-by":"crossref","unstructured":"Russakovsky, O., et al.: ICMLNet large scale visual recognition challenge. Int. J. Comput. Vis. (IJCV) (2015)","DOI":"10.1007\/s11263-015-0816-y"},{"key":"20_CR39","unstructured":"Sanyal, S., Neerkaje, A.T., Kaddour, J., Kumar, A., et\u00a0al.: Early weight averaging meets high learning rates for llm pre-training. In: Workshop on Advancing Neural Network Training: Computational Efficiency, Scalability, and Resource Optimization (WANT@ NeurIPS 2023) (2023)"},{"key":"20_CR40","unstructured":"Si, D., Yun, C.: Practical sharpness-aware minimization cannot converge all the way to optima. In: Advances in Neural Information Processing Systems (NeurIPS) (2023)"},{"key":"20_CR41","unstructured":"Smith, S., Elsen, E., De, S.: On the generalization benefit of noise in stochastic gradient descent. In: International Conference on Machine Learning (ICML) (2020)"},{"key":"20_CR42","doi-asserted-by":"crossref","unstructured":"Suzuki, K., Matsuzawa, T.: Model soups for various training and validation data. AI (2022)","DOI":"10.3390\/ai3040048"},{"key":"20_CR43","doi-asserted-by":"crossref","unstructured":"Szegedy, C., Vanhoucke, V., Ioffe, S., Shlens, J., Wojna, Z.: Rethinking the inception architecture for computer vision. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2016)","DOI":"10.1109\/CVPR.2016.308"},{"key":"20_CR44","doi-asserted-by":"crossref","unstructured":"Tseng, P., Yun, S.: Block-coordinate gradient descent method for linearly constrained nonsmooth separable optimization. J. Optim. Theory Appl. (2009)","DOI":"10.1007\/s10957-008-9458-3"},{"key":"20_CR45","unstructured":"Wortsman, M., Horton, M., Guestrin, C., Farhadi, A., Rastegari, M.: Learning neural network subspaces. In: International Conference on Machine Learning (ICML) (2021)"},{"key":"20_CR46","unstructured":"Wortsman, M., et\u00a0al.: Model soups: averaging weights of multiple fine-tuned models improves accuracy without increasing inference time. In: International Conference on Machine Learning (ICML) (2022)"},{"key":"20_CR47","doi-asserted-by":"crossref","unstructured":"Wright, S.J.: Coordinate descent algorithms. Math. Program. (2015)","DOI":"10.1007\/s10107-015-0892-3"},{"key":"20_CR48","doi-asserted-by":"crossref","unstructured":"Yin, L., Liu, S., Fang, M., Huang, T., Menkovski, V., Pechenizkiy, M.: Lottery pools: winning more by interpolating tickets without increasing training or inference cost. In: Proceedings of the AAAI Conference on Artificial Intelligence (2023)","DOI":"10.1609\/aaai.v37i9.26297"},{"key":"20_CR49","unstructured":"Yosinski, J., Clune, J., Bengio, Y., Lipson, H.: How transferable are features in deep neural networks? In: Advances in Neural Information Processing Systems (NeurIPS) (2014)"},{"key":"20_CR50","unstructured":"Yu, L., et al.: Metamath: bootstrap your own mathematical questions for large language models. In: International Conference on Learning Representations (ICLR) (2024)"},{"key":"20_CR51","doi-asserted-by":"crossref","unstructured":"Zhou, Z.H.: Ensemble Methods: Foundations and Algorithms. CRC Press, Boca Raton (2012)","DOI":"10.1201\/b12207"},{"key":"20_CR52","unstructured":"Zimmer, M., Spiegel, C., Pokutta, S.: Sparse model soups: a recipe for improved pruning via model averaging. In: International Conference on Learning Representations (ICLR) (2024)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-73650-6_20","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,20]],"date-time":"2024-11-20T19:07:25Z","timestamp":1732129645000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-73650-6_20"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,21]]},"ISBN":["9783031736490","9783031736506"],"references-count":52,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-73650-6_20","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,21]]},"assertion":[{"value":"21 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}