{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T21:02:14Z","timestamp":1784408534612,"version":"3.55.0"},"publisher-location":"Singapore","reference-count":25,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819234028","type":"print"},{"value":"9789819234035","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-3403-5_23","type":"book-chapter","created":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T20:08:35Z","timestamp":1784405315000},"page":"295-307","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Combating Modality Imbalance via Gradient Variance-Guided Tuning"],"prefix":"10.1007","author":[{"given":"Xingyue","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiansheng","family":"Fang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Na","family":"Zeng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Junjie","family":"Lin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiaqi","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hongyu","family":"Lin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiang","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"23_CR1","first-page":"609","volume-title":"Proceedings of the IEEE International Conference on Computer Vision","author":"R Arandjelovic","year":"2017","unstructured":"Arandjelovic, R., Zisserman, A.: Look, listen and learn. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 609\u2013617 (2017)"},{"issue":"4","key":"23_CR2","doi-asserted-by":"publisher","first-page":"377","DOI":"10.1109\/TAFFC.2014.2336244","volume":"5","author":"H Cao","year":"2014","unstructured":"Cao, H., et al.: Crema-d: crowd-sourced emotional multimodal actors dataset. IEEE Trans. Affect. Comput. 5(4), 377\u2013390 (2014)","journal-title":"IEEE Trans. Affect. Comput."},{"key":"23_CR3","unstructured":"Du, C., et al.: Improving multi-modal learning with Uni-modal teachers. arXiv preprint https:\/\/arxiv.org\/abs\/2106.11059. (2021)"},{"key":"23_CR4","unstructured":"Du, C. et al.: On Uni-modal feature learning in supervised multi-modal learning. In: International Conference on Machine Learning. pp. 8632\u20138656 PMLR (2023)."},{"key":"23_CR5","first-page":"20029","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Y Fan","year":"2023","unstructured":"Fan, Y., et al.: PMR: prototypical modal rebalance for multimodal learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 20029\u201320038 (2023)"},{"key":"23_CR6","first-page":"770","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","author":"K He","year":"2016","unstructured":"He, K., et al.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)"},{"key":"23_CR7","doi-asserted-by":"publisher","first-page":"61","DOI":"10.1145\/3551876.3554811","volume-title":"Proceedings of the 3rd International on Multimodal Sentiment Analysis Workshop and Challenge","author":"Y He","year":"2022","unstructured":"He, Y., et al.: Multimodal temporal attention in sentiment analysis. In: Proceedings of the 3rd International on Multimodal Sentiment Analysis Workshop and Challenge, pp. 61\u201366 (2022)"},{"key":"23_CR8","first-page":"25854","volume-title":"Proceedings of the Computer Vision and Pattern Recognition Conference (CVPR)","author":"C Huang","year":"2025","unstructured":"Huang, C., et al.: Adaptive unimodal regulation for balanced multimodal information acquisition. In: Proceedings of the Computer Vision and Pattern Recognition Conference (CVPR), pp. 25854\u201325863 (2025)"},{"key":"23_CR9","unstructured":"Huang, Y. et al.: Modality competition: what makes joint training of multi-modal network fail in deep learning? (provably). In: International Conference on Machine Learning. pp. 9226\u20139259 PMLR (2022)."},{"issue":"11","key":"23_CR10","doi-asserted-by":"publisher","first-page":"3137","DOI":"10.1109\/TMM.2018.2823900","volume":"20","author":"Y-G Jiang","year":"2018","unstructured":"Jiang, Y.-G., et al.: Modeling multimodal clues in a hybrid deep learning framework for video classification. IEEE Trans. Multimed. 20(11), 3137\u20133147 (2018)","journal-title":"IEEE Trans. Multimed."},{"issue":"1","key":"23_CR11","first-page":"5198","volume":"32","author":"D Kiela","year":"2018","unstructured":"Kiela, D., et al.: Efficient large-scale multi-modal classification. Proc. AAAI Conf. Artif. Intell. 32(1), 5198\u20135204 (2018)","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"key":"23_CR12","unstructured":"Lin, B., et al.: Variational probabilistic fusion network for RGB-t semantic segmentation. arXiv preprint https:\/\/arxiv.org\/abs\/2307.08536. (2023)"},{"key":"23_CR13","first-page":"211","volume-title":"Proceedings of the IEEE\/CVF conference on Computer Vision and Pattern Recognition","author":"X Lin","year":"2024","unstructured":"Lin, X., et al.: Suppress and rebalance: towards generalized multi-modal face anti-spoofing. In: Proceedings of the IEEE\/CVF conference on Computer Vision and Pattern Recognition, pp. 211\u2013221. IEEE, Piscataway (2024)"},{"key":"23_CR14","first-page":"689","volume-title":"Proceedings of the 28th International Conference on Machine Learning (ICML-11)","author":"J Ngiam","year":"2011","unstructured":"Ngiam, J., et al.: Multimodal deep learning. In: Proceedings of the 28th International Conference on Machine Learning (ICML-11), pp. 689\u2013696. Omnipress (2011)"},{"key":"23_CR15","first-page":"8238","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"X Peng","year":"2022","unstructured":"Peng, X., et al.: Balanced multimodal learning via on-the-fly gradient modulation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8238\u20138247. IEEE, Piscataway (2022)"},{"issue":"1","key":"23_CR16","first-page":"3942","volume":"32","author":"E Perez","year":"2018","unstructured":"Perez, E., et al.: Film: visual reasoning with a general conditioning layer. Proc. AAAI Conf. Artif. Intell. 32(1), 3942\u20133951 (2018)","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"key":"23_CR17","first-page":"11","volume":"9","author":"L Van der Maaten","year":"2008","unstructured":"Van der Maaten, L., Hinton, G.: Visualizing data using T-SNE. J. Mach. Learn. Res. 9, 11 (2008)","journal-title":"J. Mach. Learn. Res."},{"key":"23_CR18","first-page":"12695","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"W Wang","year":"2020","unstructured":"Wang, W., et al.: What makes training multi-modal classification networks hard? In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12695\u201312705. IEEE, Piscataway (2020)"},{"key":"23_CR19","first-page":"27338","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Y Wei","year":"2024","unstructured":"Wei, Y., et al.: Enhancing multimodal cooperation via sample-level modality valuation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 27338\u201327347. IEEE, Piscataway (2024)"},{"key":"23_CR20","doi-asserted-by":"publisher","DOI":"10.1016\/j.media.2023.102938","volume":"90","author":"J Wu","year":"2023","unstructured":"Wu, J., et al.: Gamma challenge: glaucoma grading from multi-modality images. Med. Image Anal. 90, 102938 (2023)","journal-title":"Med. Image Anal."},{"key":"23_CR21","unstructured":"Wu, N. et al.: Characterizing and overcoming the greedy nature of learning in multi-modal deep neural networks. In: International Conference on Machine Learning. 162 pp. 24043\u201324055 PMLR (2022)."},{"key":"23_CR22","first-page":"19989","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Y Xia","year":"2022","unstructured":"Xia, Y., Zhao, Z.: Cross-modal background suppression for audio-visual event localization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 19989\u201319998. IEEE, Piscataway (2022)"},{"key":"23_CR23","unstructured":"Xiao, F., et al.: Audiovisual slow fast networks for video recognition. arXiv preprint https:\/\/arxiv.org\/abs\/2001.08740 (2020)"},{"key":"23_CR24","doi-asserted-by":"publisher","first-page":"18694","DOI":"10.52202\/068431-1358","volume":"35","author":"Z Xue","year":"2022","unstructured":"Xue, Z., et al.: Large-batch optimization for dense visual predictions: training faster R-CNN in 4.2 minutes. Adv. Neural Inf. Process. Syst. 35, 18694\u201318706 (2022)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"23_CR25","first-page":"27456","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"X Zhang","year":"2024","unstructured":"Zhang, X., et al.: Multimodal representation learning by alternating unimodal adaptation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 27456\u201327466. IEEE, Piscataway (2024)"}],"container-title":["Lecture Notes in Computer Science","Advanced Intelligent Computing Technology and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-3403-5_23","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T20:08:38Z","timestamp":1784405318000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-3403-5_23"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"ISBN":["9789819234028","9789819234035"],"references-count":25,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-3403-5_23","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"19 July 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICIC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Intelligent Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Toronto","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Canada","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icic2026a","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/www.ic-icc.cn\/2026\/index.htm","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}