{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,8]],"date-time":"2026-06-08T10:57:44Z","timestamp":1780916264168,"version":"3.54.1"},"publisher-location":"Singapore","reference-count":29,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819500130","type":"print"},{"value":"9789819500147","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-95-0014-7_16","type":"book-chapter","created":{"date-parts":[[2025,7,24]],"date-time":"2025-07-24T10:05:49Z","timestamp":1753351549000},"page":"185-196","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["SViQA: A Unified Speech-Vision Multimodal Model for Textless Visual Question Answering"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-5293-3926","authenticated-orcid":false,"given":"Bingxin","family":"Li","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,7,25]]},"reference":[{"key":"16_CR1","first-page":"34892","volume":"36","author":"H Liu","year":"2023","unstructured":"Liu, H., Li, C., Wu, Q., Lee, Y.J.: Visual instruction tuning. Adv. Neural. Inf. Process. Syst. 36, 34892\u201334916 (2023)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"issue":"2","key":"16_CR2","doi-asserted-by":"publisher","first-page":"423","DOI":"10.1109\/TPAMI.2018.2798607","volume":"41","author":"T Baltru\u0161aitis","year":"2018","unstructured":"Baltru\u0161aitis, T., Ahuja, C., Morency, L.P.: Multimodal machine learning: A survey and taxonomy. IEEE Trans. Pattern Anal. Mach. Intell. 41(2), 423\u2013443 (2018)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"16_CR3","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., et al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning. pp. 8748\u20138763. PmLR (2021)"},{"key":"16_CR4","doi-asserted-by":"crossref","unstructured":"Guzhov, A., Raue, F., Hees, J., Dengel, A.: Audioclip: extending clip to image, text and audio. In: ICASSP 2022\u20132022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). pp. 976\u2013980. IEEE (2022)","DOI":"10.1109\/ICASSP43922.2022.9747631"},{"key":"16_CR5","doi-asserted-by":"crossref","unstructured":"Wu, H.H., Seetharaman, P., Kumar, K., Bello, J.P.: Wav2clip: learning robust audio representations from clip. In: ICASSP 2022\u20132022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). pp. 4563\u20134567. IEEE (2022)","DOI":"10.1109\/ICASSP43922.2022.9747669"},{"key":"16_CR6","doi-asserted-by":"crossref","unstructured":"Oneat, \u0103, D., Cucu, H.: Improving multimodal speech recognition by data augmentation and speech representations. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 4579\u20134588 (2022)","DOI":"10.1109\/CVPRW56347.2022.00504"},{"key":"16_CR7","doi-asserted-by":"crossref","unstructured":"Huang, R., Li, M., Yang, D., Shi, J., Chang, X., Ye, Z., Wu, Y., Hong, Z., Huang, J., Liu, J., et al.: Audiogpt: understanding and generating speech, music, sound, and talking head. Proc. AAAI Conf. Artif. Intell. 38, 23802\u201323804 (2024)","DOI":"10.1609\/aaai.v38i21.30570"},{"key":"16_CR8","doi-asserted-by":"crossref","unstructured":"Reddy, V.M., Vaishnavi, T., Kumar, K.P.: Speech-to-text and text-to-speech recognition using deep learning. In: 2023 2nd International Conference on Edge Computing and Applications (ICECAA). pp. 657\u2013666. IEEE (2023)","DOI":"10.1109\/ICECAA58104.2023.10212222"},{"key":"16_CR9","doi-asserted-by":"crossref","unstructured":"Zhang, D., et al.: Speechgpt: empowering large language models with intrinsic cross-modal conversational abilities (2023). arXiv preprint arXiv:2305.11000","DOI":"10.18653\/v1\/2023.findings-emnlp.1055"},{"key":"16_CR10","unstructured":"Fu, C., Lin, H., Long, Z., Shen, Y., Zhao, M., Zhang, Y., Dong, S., Wang, X., Yin, D., Ma, L., et al.: Vita: towards open-source interactive omni multimodal llm (2024). arXiv preprint arXiv:2408.05211"},{"key":"16_CR11","unstructured":"Fathullah, Y., et al.: Audiochatllama: towards general-purpose speech abilities for llms (2023). arXiv preprint arXiv:2311.06753"},{"key":"16_CR12","doi-asserted-by":"crossref","unstructured":"Lyu, Y., Zheng, X., Zhou, J., Wang, L.: Unibind: llm-augmented unified and balanced representation space to bind them all. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 26752\u201326762 (2024)","DOI":"10.1109\/CVPR52733.2024.02526"},{"key":"16_CR13","doi-asserted-by":"crossref","unstructured":"Zhan, J., Dai, J., Ye, J., Zhou, Y., Zhang, D., Liu, Z., Zhang, X., Yuan, R., Zhang, G., Li, L., et al.: Anygpt: unified multimodal llm with discrete sequence modelling (2024). arXiv preprint arXiv:2402.12226","DOI":"10.18653\/v1\/2024.acl-long.521"},{"key":"16_CR14","unstructured":"Wu, S., Fei, H., Qu, L., Ji, W., Chua, T.S.: Next-gpt: any-to-any multimodal llm. In: Forty-first International Conference on Machine Learning (2024)"},{"key":"16_CR15","doi-asserted-by":"crossref","unstructured":"Antol, S., et al.: Vqa: visual question answering. In: Proceedings of the IEEE International Conference on Computer Vision. pp. 2425\u20132433 (2015)","DOI":"10.1109\/ICCV.2015.279"},{"key":"16_CR16","doi-asserted-by":"crossref","unstructured":"Tan, H., Bansal, M.: Lxmert: learning cross-modality encoder representations from transformers (2019). arXiv preprint arXiv:1908.07490","DOI":"10.18653\/v1\/D19-1514"},{"key":"16_CR17","unstructured":"Kim, W., Son, B., Kim, I.: Vilt: vision-and-language transformer without convolution or region supervision. In: International Conference on Machine Learning. pp. 5583\u20135594. PMLR (2021)"},{"key":"16_CR18","first-page":"9617","volume":"35","author":"Z Tang","year":"2022","unstructured":"Tang, Z., Cho, J., Nie, Y., Bansal, M.: Tvlt: textless vision-language transformer. Adv. Neural. Inf. Process. Syst. 35, 9617\u20139632 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"16_CR19","doi-asserted-by":"crossref","unstructured":"Alasmary, F., Al-Ahmadi, S.: Sbvqa 2.0: robust end-to-end speech-based visual question answering for open-ended questions. IEEE Access 11, 140967\u2013140980 (2023)","DOI":"10.1109\/ACCESS.2023.3339537"},{"key":"16_CR20","unstructured":"Rubenstein, P.K., Asawaroengchai, C., Nguyen, D.D., Bapna, A., Borsos, Z., Quitry, F.d.C., Chen, P., Badawy, D.E., Han, W., Kharitonov, E., et al.: Audiopalm: a large language model that can speak and listen (2023). arXiv preprint arXiv:2306.12925"},{"key":"16_CR21","unstructured":"Wang, T., et al.: Viola: unified codec language models for speech recognition, synthesis, and translation (2023). arXiv preprint arXiv:2305.16107"},{"key":"16_CR22","unstructured":"Shu, Y., et al.: Llasm: large language and speech model (2023). arXiv preprint arXiv:2308.15930"},{"key":"16_CR23","unstructured":"Tang, C., et al.: Salmonn: Towards generic hearing abilities for large language models (2023). arXiv preprint arXiv:2310.13289"},{"key":"16_CR24","unstructured":"Fang, Q., Guo, S., Zhou, Y., Ma, Z., Zhang, S., Feng, Y.: Llama-omni: Seamless speech interaction with large language models (2024). arXiv preprint arXiv:2409.06666"},{"key":"16_CR25","unstructured":"Xie, Z., Wu, C.: Mini-omni2: Towards open-source gpt-4o with vision, speech and duplex capabilities (2024). arXiv preprint arXiv:2410.11190"},{"key":"16_CR26","doi-asserted-by":"crossref","unstructured":"Laput, G.P., et al.: Pixeltone: a multimodal interface for image editing. In: Proceedings of the SIGCHI Conference on Human Factors in Computing Systems. pp. 2185\u20132194 (2013)","DOI":"10.1145\/2470654.2481301"},{"key":"16_CR27","doi-asserted-by":"crossref","unstructured":"Wang, X., Qiao, T., Zhu, J., Hanjalic, A., Scharenborg, O.: S2igan: speech-to-image generation via adversarial learning (2020). arXiv preprint arXiv:2005.06968","DOI":"10.21437\/Interspeech.2020-1759"},{"key":"16_CR28","unstructured":"Zhang, T., Dai, D., Tuytelaars, T., Moens, M.F., Van Gool, L.: Speech-based visual question answering (2017). arXiv preprint arXiv:1705.00464"},{"key":"16_CR29","unstructured":"Zhou, B., et al.: Tinyllava: a framework of small-scale large multimodal models (2024). arXiv preprint arXiv:2402.14289"}],"container-title":["Lecture Notes in Computer Science","Advanced Intelligent Computing Technology and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-95-0014-7_16","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,8]],"date-time":"2026-06-08T09:57:49Z","timestamp":1780912669000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-95-0014-7_16"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9789819500130","9789819500147"],"references-count":29,"URL":"https:\/\/doi.org\/10.1007\/978-981-95-0014-7_16","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"25 July 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICIC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Intelligent Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Ningbo","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 July 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 July 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"21","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icic2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/www.ic-icc.cn\/icg\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}