{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T08:02:40Z","timestamp":1784361760571,"version":"3.55.0"},"publisher-location":"Singapore","reference-count":27,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819234288","type":"print"},{"value":"9789819234295","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-3429-5_11","type":"book-chapter","created":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T07:06:49Z","timestamp":1784358409000},"page":"127-138","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Semantic-Guided Visual Byte-Pair Encoding for Unified Autoregressive Multimodal Modeling"],"prefix":"10.1007","author":[{"given":"Kangwei","family":"Liu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"WenLong","family":"Dong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lijian","family":"Gao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qirong","family":"Mao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"issue":"2","key":"11_CR1","doi-asserted-by":"publisher","first-page":"423","DOI":"10.1109\/TPAMI.2018.2798607","volume":"41","author":"T Baltru\u0161aitis","year":"2018","unstructured":"Baltru\u0161aitis, T., Ahuja, C., Morency, L.-P.: Multimodal machine learning: a survey and taxonomy. IEEE Trans. Pattern Anal. Mach. Intell. 41(2), 423\u2013443 (2018)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"11_CR2","unstructured":"Chen, G.H., Chen, S., Zhang, R., Wan, X., Wang, B.: ALLaVA: Harnessing GPT4V-synthesized data for a lite vision-language model. arXiv preprint arXiv:2402.11684 (2024)"},{"key":"11_CR3","unstructured":"Fu, C., et al.: MME: A comprehensive evaluation benchmark for multimodal large language models. In: The 39th Annual Conference on Neural Information Processing Systems Datasets and Benchmarks Track (2025)"},{"key":"11_CR4","doi-asserted-by":"crossref","unstructured":"Goyal, Y., Khot, T., Summers-Stay, D., Batra, D., Parikh, D.: Making the V in VQA matter: elevating the role of image understanding in visual question answering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6904\u20136913 (2017)","DOI":"10.1109\/CVPR.2017.670"},{"key":"11_CR5","doi-asserted-by":"crossref","unstructured":"Gurari, D., et al.: VizWiz Grand Challenge: answering visual questions from blind people. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3608\u20133617 (2018)","DOI":"10.1109\/CVPR.2018.00380"},{"key":"11_CR6","doi-asserted-by":"publisher","first-page":"787","DOI":"10.3115\/v1\/D14-1086","volume-title":"Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing (EMNLP)","author":"S Kazemzadeh","year":"2014","unstructured":"Kazemzadeh, S., Ordonez, V., Matten, M., Berg, T.: ReferItGame: Referring to objects in photographs of natural scenes. In: Moschitti, A., Pang, B., Daelemans, W. (eds.) Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing (EMNLP), pp. 787\u2013798. Association for Computational Linguistics, Doha, Qatar (2014)"},{"key":"11_CR7","unstructured":"Li, B., et al.: MIMIC-IT: Multi-modal in-context instruction tuning. arXiv preprint arXiv:2306.05425 (2023)"},{"key":"11_CR8","unstructured":"Li, J., Li, D., Savarese, S., Hoi, S.: BLIP-2: Bootstrapping language-image pre-training with frozen image encoders and large language models. In: International Conference on Machine Learning, pp. 19730\u201319742. PMLR (2023)"},{"key":"11_CR9","doi-asserted-by":"crossref","unstructured":"Li, Y., Du, Y., Zhou, K., Wang, J., Zhao, W.X., Wen, J.R.: Evaluating object hallucination in large vision-language models. arXiv preprint arXiv:2305.10355 (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.20"},{"key":"11_CR10","doi-asserted-by":"crossref","unstructured":"Li, Z., Wang, W., Cai, Y., Xu, Q., Wang, P., Wang, T.: UnifiedMLLM: Enabling unified representation for multi-modal multi-tasks with large language model. In: Findings of the Association for Computational Linguistics: NAACL 2025, pp. 334\u2013344 (2025)","DOI":"10.18653\/v1\/2025.findings-naacl.19"},{"key":"11_CR11","doi-asserted-by":"crossref","unstructured":"Lin, B., et al.: Video-LLaVA: Learning united visual representation by alignment before projection. In: Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing, pp. 5971\u20135984 (2024)","DOI":"10.18653\/v1\/2024.emnlp-main.342"},{"key":"11_CR12","doi-asserted-by":"crossref","unstructured":"Liu, H., Li, C., Li, Y., Lee, Y.J.: Improved baselines with visual instruction tuning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 26296\u201326306 (2024)","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"11_CR13","doi-asserted-by":"crossref","unstructured":"Liu, Y., et al.: MMBench: Is your multi-modal model an all-around player? In: European Conference on Computer Vision, pp. 216\u2013233. Springer (2024)","DOI":"10.1007\/978-3-031-72658-3_13"},{"key":"11_CR14","unstructured":"Ping, B., et al.: PaCo-RL: Advancing reinforcement learning for consistent image generation with pairwise reward modeling. arXiv preprint arXiv:2512.04784 (2025)"},{"key":"11_CR15","doi-asserted-by":"crossref","unstructured":"Schwenk, D., Khandelwal, A., Clark, C., Marino, K., Mottaghi, R.: A-OKVQA: A benchmark for visual question answering using world knowledge. In: European Conference on Computer Vision, pp. 146\u2013162. Springer (2022)","DOI":"10.1007\/978-3-031-20074-8_9"},{"key":"11_CR16","doi-asserted-by":"crossref","unstructured":"Sharma, P., Soricut, R.: Conceptual Captions: a cleaned, hypernymed, image alt-text dataset for automatic image captioning. In: Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 2556\u20132565 (2018)","DOI":"10.18653\/v1\/P18-1238"},{"key":"11_CR17","unstructured":"Chameleon Team: Chameleon: Mixed-modal early-fusion foundation models. arXiv preprint arXiv:2405.09818 (2024)"},{"key":"11_CR18","unstructured":"Qwen Team: Qwen3 Technical Report. arXiv preprint arXiv:2505.09388 (2025)"},{"key":"11_CR19","unstructured":"Vaswani, A., et al.: Attention is all you need. Advances in Neural Information Processing Systems 30 (NIPS 2017) (2017)"},{"key":"11_CR20","doi-asserted-by":"crossref","unstructured":"Wang, Z.M., et al.: MIO: A foundation model on multimodal tokens. In: Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing, pp. 5077\u20135099 (2025)","DOI":"10.18653\/v1\/2025.emnlp-main.255"},{"issue":"2","key":"11_CR21","doi-asserted-by":"publisher","DOI":"10.1007\/s11432-024-4222-0","volume":"68","author":"Z Xi","year":"2025","unstructured":"Xi, Z., et al.: The rise and potential of large language model based agents: a survey. Sci. China Inf. Sci. 68(2), 121101 (2025)","journal-title":"Sci. China Inf. Sci."},{"key":"11_CR22","unstructured":"Ye, Q., et al.: mPLUG-Owl: Modularization empowers large language models with multimodality. arXiv preprint arXiv:2304.14178 (2023)"},{"issue":"12","key":"11_CR23","doi-asserted-by":"publisher","DOI":"10.1093\/nsr\/nwae403","volume":"11","author":"S Yin","year":"2024","unstructured":"Yin, S., et al.: A survey on multimodal large language models. Natl. Sci. Rev. 11(12), nwae403 (2024)","journal-title":"Natl. Sci. Rev."},{"key":"11_CR24","doi-asserted-by":"crossref","unstructured":"Zhan, J., et al.: AnyGPT: Unified multimodal LLM with discrete sequence modeling. In: Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 9637\u20139662 (2024)","DOI":"10.18653\/v1\/2024.acl-long.521"},{"key":"11_CR25","unstructured":"Zhang, J., et al.: ProVision: Programmatically scaling vision-centric instruction data for multimodal language models. arXiv preprint arXiv:2412.07012 (2024)"},{"key":"11_CR26","doi-asserted-by":"crossref","unstructured":"Zhang, W., Feng, Y., Lu, Z.: Unified multimodal understanding via byte-pair visual encoding. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (2025)","DOI":"10.1109\/ICCV51701.2025.01206"},{"key":"11_CR27","unstructured":"Zhang, W., et al.: From pixels to tokens: Byte-pair encoding on quantized visual modalities. arXiv preprint arXiv:2410.02155 (2024)"}],"container-title":["Lecture Notes in Computer Science","Advanced Intelligent Computing Technology and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-3429-5_11","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T07:06:52Z","timestamp":1784358412000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-3429-5_11"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"ISBN":["9789819234288","9789819234295"],"references-count":27,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-3429-5_11","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"19 July 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICIC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Intelligent Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Toronto","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Canada","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icic2026a","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/www.ic-icc.cn\/2026\/index.htm","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}