{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T08:03:37Z","timestamp":1784189017973,"version":"3.55.0"},"publisher-location":"Singapore","reference-count":16,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819234165","type":"print"},{"value":"9789819234172","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T00:00:00Z","timestamp":1784246400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T00:00:00Z","timestamp":1784246400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-3417-2_6","type":"book-chapter","created":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T07:09:43Z","timestamp":1784185783000},"page":"61-70","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Sound-Mind: Enhancing Paralinguistic Understanding in MLLMs via Iterative Latent Refinement"],"prefix":"10.1007","author":[{"given":"Zongzheng","family":"Han","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xuwen","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wei","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zehua","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jueting","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tingting","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ziyang","family":"Xing","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zongjian","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,17]]},"reference":[{"key":"6_CR1","unstructured":"Chu, Y., et al.: Qwen2-audio technical report. arXiv preprint arXiv:2407.10759. (2024)"},{"key":"6_CR2","volume-title":"Proceedings of ICLR","author":"C Tang","year":"2024","unstructured":"Tang, C., et al.: SALMONN: towards generic hearing abilities for large language models. In: Proceedings of ICLR (2024)"},{"key":"6_CR3","volume-title":"Proc. ICLR","author":"S Hao","year":"2024","unstructured":"Hao, S., et al.: Training large language models to reason in a continuous latent space. In: Proc. ICLR (2024)"},{"issue":"2","key":"6_CR4","first-page":"3","volume":"1","author":"EJ Hu","year":"2022","unstructured":"Hu, E.J., et al.: Lora: low-rank adaptation of large language models. ICLR. 1(2), 3 (2022)","journal-title":"ICLR"},{"key":"6_CR5","doi-asserted-by":"publisher","first-page":"527","DOI":"10.18653\/v1\/P19-1050","volume-title":"Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics","author":"S Poria","year":"2019","unstructured":"Poria, S., Hazarika, D., Majumder, N., Naik, G., Cambria, E., Mihalcea, R.: Meld: a multimodal multi-party dataset for emotion recognition in conversations. In: Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics, pp. 527\u2013536 (2019)"},{"key":"6_CR6","doi-asserted-by":"crossref","unstructured":"Schuller, B., Steidl, S., Batliner, A.: The interspeech 2009 emotion challenge (2009)","DOI":"10.21437\/Interspeech.2009-103"},{"issue":"6","key":"6_CR7","doi-asserted-by":"publisher","first-page":"1179","DOI":"10.1109\/JSTSP.2022.3207050","volume":"16","author":"A Mohamed","year":"2022","unstructured":"Mohamed, A., et al.: Self-supervised speech representation learning: a review. IEEE J. Sel. Top. Sign. Proces. 16(6), 1179\u20131210 (2022)","journal-title":"IEEE J. Sel. Top. Sign. Proces."},{"key":"6_CR8","doi-asserted-by":"publisher","first-page":"3451","DOI":"10.1109\/TASLP.2021.3122291","volume":"29","author":"WN Hsu","year":"2021","unstructured":"Hsu, W.N., Bolte, B., Tsai, Y.H.H., Lakhotia, K., Mohamed, A.: Hubert: self-supervised speech representation learning by masked prediction of hidden units. IEEE\/ACM Trans. Audio Speech Lang. Process. 29, 3451\u20133460 (2021)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"issue":"8","key":"6_CR9","first-page":"1","volume":"58","author":"E Yang","year":"2026","unstructured":"Yang, E., et al.: Model merging in LLMs, MLLMs, and beyond: methods, theories, applications, and opportunities. ACM Comput. Surv. 58(8), 1\u201341 (2026)","journal-title":"ACM Comput. Surv."},{"key":"6_CR10","doi-asserted-by":"publisher","first-page":"24824","DOI":"10.52202\/068431-1800","volume":"35","author":"J Wei","year":"2022","unstructured":"Wei, J., et al.: Chain-of-thought prompting elicits reasoning in large language models. Adv. Neural Inf. Proces. Syst. 35, 24824\u201324837 (2022)","journal-title":"Adv. Neural Inf. Proces. Syst."},{"key":"6_CR11","doi-asserted-by":"publisher","first-page":"23840","DOI":"10.18653\/v1\/2025.emnlp-main.1216","volume-title":"Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing","author":"X Zhifei","year":"2025","unstructured":"Zhifei, X., Lin, M., Liu, Z., Wu, P., Yan, S., Miao, C.: Audio-reasoner: improving reasoning capability in large audio language models. In: Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing, pp. 23840\u201323862 (2025)"},{"issue":"8","key":"6_CR12","doi-asserted-by":"publisher","first-page":"227","DOI":"10.1007\/s10462-025-11236-4","volume":"58","author":"L Wang","year":"2025","unstructured":"Wang, L., et al.: Parameter-efficient fine-tuning in large language models: a survey of methodologies. Artif. Intell. Rev. 58(8), 227 (2025)","journal-title":"Artif. Intell. Rev."},{"key":"6_CR13","first-page":"28492","volume-title":"International Conference on Machine Learning","author":"A Radford","year":"2023","unstructured":"Radford, A., Kim, J.W., Xu, T., Brockman, G., McLeavey, C., Sutskever, I.: Robust speech recognition via large-scale weak supervision. In: International Conference on Machine Learning, pp. 28492\u201328518. PMLR (2023)"},{"key":"6_CR14","unstructured":"Liu, Y., et al.: Roberta: a robustly optimized BERT pretraining approach. arXiv preprint arXiv: 1907.11692. (2019)"},{"issue":"9","key":"6_CR15","doi-asserted-by":"publisher","first-page":"10745","DOI":"10.1109\/TPAMI.2023.3263585","volume":"45","author":"J Wagner","year":"2023","unstructured":"Wagner, J., et al.: Dawn of the transformer era in speech emotion recognition: closing the valence gap. IEEE Trans. Pattern Anal. Mach. Intell. 45(9), 10745\u201310759 (2023)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"6_CR16","doi-asserted-by":"publisher","first-page":"2523","DOI":"10.1109\/TASLP.2023.3288409","volume":"31","author":"Z Borsos","year":"2023","unstructured":"Borsos, Z., et al.: AudioLM: a language modeling approach to audio generation. IEEE\/ACM Trans. Audio Speech Lang. Process. 31, 2523\u20132533 (2023)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."}],"container-title":["Lecture Notes in Computer Science","Advanced Intelligent Computing Technology and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-3417-2_6","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T07:09:46Z","timestamp":1784185786000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-3417-2_6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,17]]},"ISBN":["9789819234165","9789819234172"],"references-count":16,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-3417-2_6","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7,17]]},"assertion":[{"value":"17 July 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICIC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Intelligent Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Toronto","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Canada","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icic2026a","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/www.ic-icc.cn\/2026\/index.htm","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}