{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,23]],"date-time":"2026-05-23T09:08:48Z","timestamp":1779527328930,"version":"3.53.1"},"reference-count":47,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2026,4,13]],"date-time":"2026-04-13T00:00:00Z","timestamp":1776038400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,4,13]],"date-time":"2026-04-13T00:00:00Z","timestamp":1776038400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100018537","name":"National Science and Technology Major Project","doi-asserted-by":"publisher","award":["2023ZD0121102"],"award-info":[{"award-number":["2023ZD0121102"]}],"id":[{"id":"10.13039\/501100018537","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2026,5]]},"DOI":"10.1007\/s11263-026-02840-0","type":"journal-article","created":{"date-parts":[[2026,4,13]],"date-time":"2026-04-13T03:02:16Z","timestamp":1776049336000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["AdViP: Aligning Multi-modal LLMs via Adaptive Vision-enhanced Preference Optimization"],"prefix":"10.1007","volume":"134","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-7322-9942","authenticated-orcid":false,"given":"Jinda","family":"Lu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jinghan","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuan","family":"Gao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Junkang","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiancan","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiang","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiangnan","family":"He","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,4,13]]},"reference":[{"key":"2840_CR1","unstructured":"Bai, J., Bai, S., Yang, S., Wang, S., Tan, S., Wang, P., Lin, J., Zhou, C., & Zhou, J. (2023). Qwen-vl: A versatile vision-language model for understanding, localization, text reading, and beyond. arXiv preprint arXiv:2308.12966, 1(2):3."},{"key":"2840_CR2","doi-asserted-by":"crossref","unstructured":"Bradley, R. A., & Terry, M. E. (1952). Rank analysis of incomplete block designs: I. the method of paired comparisons. Biometrika, 39(3\/4):324\u2013345.","DOI":"10.1093\/biomet\/39.3-4.324"},{"key":"2840_CR3","doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., & Zagoruyko, S. (2020). End-to-end object detection with transformers. In European conference on computer vision, pages 213\u2013229. Springer.","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"2840_CR4","doi-asserted-by":"crossref","unstructured":"Chen, Y., Tan, J., Zhang, A., Yang, Z., Sheng, L., Zhang, E., Wang, X., & Chua, T. S. (2024a). On softmax direct preference optimization for recommendation. arXiv preprint arXiv:2406.09215 .","DOI":"10.52202\/079017-0863"},{"key":"2840_CR5","doi-asserted-by":"crossref","unstructured":"Chen, Z., Wu, J., Wang, W., Su, W., Chen, G., Xing, S., Zhong, M., Zhang, Q., Zhu, X., Lu, L., and others. (2024b). Internvl: Scaling up vision foundation models and aligning for generic visual-linguistic tasks. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pages 24185\u201324198.","DOI":"10.1109\/CVPR52733.2024.02283"},{"key":"2840_CR6","unstructured":"Cui, C., Zhang, A., Zhou, Y., Chen, Z., Deng, G., Yao, H., & Chua, T. S. (2024). Fine-grained verifiers: Preference modeling as next-token prediction in vision-language alignment. arXiv preprint arXiv:2410.14148."},{"key":"2840_CR7","unstructured":"Fu, C., Chen, P., Shen, Y., Qin, Y., Zhang, M., Lin, X., Yang, J., Zheng, X., Li, K., Sun, X., and others. (2025). Mme: A comprehensive evaluation benchmark for multimodal large language models. In The Thirty-ninth Annual Conference on Neural Information Processing Systems Datasets and Benchmarks Track."},{"key":"2840_CR8","unstructured":"Grattafiori, A., Dubey, A., Jauhri, A., Pandey, A., Kadian, A., Al-Dahle, A., Letman, Mathur, A., Schelten, A., Yang, A., & Fan, A., and others (2024) . The llama 3 herd of models. arXiv preprint arXiv:2407.21783."},{"key":"2840_CR9","doi-asserted-by":"crossref","unstructured":"Gu, J., Wang, Y., Cao, M., Bu, P., Song, J., He, Y., Li, S.,& Zheng, B. (2024). Token preference optimization with self-calibrated visual-anchored rewards for hallucination mitigation. arXiv preprint arXiv:2412.14487.","DOI":"10.18653\/v1\/2025.findings-emnlp.1076"},{"key":"2840_CR10","doi-asserted-by":"crossref","unstructured":"Huang, Q., Dong, X., Zhang, P., Wang, B., He, C., Wang, J., Lin, D., Zhang, W., & Yu, N. (2024). Opera: Alleviating hallucination in multi-modal large language models via over-trust penalty and retrospection-allocation. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pages 13418\u201313427.","DOI":"10.1109\/CVPR52733.2024.01274"},{"key":"2840_CR11","doi-asserted-by":"crossref","unstructured":"Jain, J., Yang, J., & Shi, H. (2024). Vcoder: Versatile vision encoders for multimodal large language models. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pages 27992\u201328002.","DOI":"10.1109\/CVPR52733.2024.02644"},{"key":"2840_CR12","doi-asserted-by":"crossref","unstructured":"Kirillov, A., Mintun, E., Ravi, N., Mao, H., Rolland, C., Gustafson, L., Xiao, T., Whitehead, S., Berg, A.C., Lo, W.-Y., and others. (2023). Segment anything. In Proceedings of the IEEE\/CVF international conference on computer vision, pages 4015\u20134026.","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"2840_CR13","doi-asserted-by":"crossref","unstructured":"Leng, S., Zhang, H., Chen, G., Li, X., Lu, S., Miao, C., & Bing, L. (2024). Mitigating object hallucinations in large vision-language models through visual contrastive decoding. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pages 13872\u201313882.","DOI":"10.1109\/CVPR52733.2024.01316"},{"key":"2840_CR14","unstructured":"Liu, H., Li, C., Wu, Q., & Lee, Y. J. (2024c). Visual instruction tuning. Advances in neural information processing systems, 36."},{"key":"2840_CR15","unstructured":"Liu, F., Lin, K., Li, L., Wang, J., Yacoob, Y., & Wang, L. (2023). Mitigating hallucination in large multi-modal models via robust instruction tuning. In The Twelfth International Conference on Learning Representations."},{"key":"2840_CR16","unstructured":"Liu, H., Xue, W., Chen, Y., Chen, D., Zhao, X., Wang, K., Hou, L., Li, R., & Peng, W. (2024a) A survey on hallucination in large vision-language models. arXiv preprint arXiv:2402.00253."},{"key":"2840_CR17","doi-asserted-by":"crossref","unstructured":"Liu, S., Zeng, Z., Ren, T., Li, F., Zhang, H., Yang, J., Jiang, Q., Li, C., Yang, J., Su, H., and others. (2024d). Grounding dino: Marrying dino with grounded pre-training for open-set object detection. In European Conference on Computer Vision, pages 38\u201355. Springer.","DOI":"10.1007\/978-3-031-72970-6_3"},{"key":"2840_CR18","volume-title":"and Yong Jae Lee","author":"H Liu","year":"2024","unstructured":"Liu, H., Li, C., Li, Y., Li, B., Zhang, Y., & Shen, S. (2024). and Yong Jae Lee. Llava-next: Improved reasoning, ocr, and world knowledge."},{"key":"2840_CR19","unstructured":"Lu, J., Wu, J., Li, J., Jia, X., Wang, S., Zhang, Y., Fang, J., Wang, X., & He, X. (2025). Dama: Data- and model-aware alignment of multi-modal llms. arXiv preprint arXiv:2502.01943."},{"key":"2840_CR20","unstructured":"Meng, Y., Xia, M., & Chen, D. (2024). Simpo: Simple preference optimization with a reference-free reward. arXiv preprint arXiv:2405.14734."},{"key":"2840_CR21","first-page":"27730","volume":"35","author":"L Ouyang","year":"2022","unstructured":"Ouyang, L., Jeffrey, W., Jiang, X., Almeida, D., Wainwright, C., Mishkin, P., Zhang, C., Agarwal, S., Slama, K., Ray, A., et al. (2022). Training language models to follow instructions with human feedback. Advances in neural information processing systems, 35, 27730\u201327744.","journal-title":"Advances in neural information processing systems"},{"issue":"2","key":"2840_CR22","first-page":"193","volume":"24","author":"RL Plackett","year":"1975","unstructured":"Plackett, R. L. (1975). The analysis of permutations. Journal of the Royal Statistical Society Series C: Applied Statistics, 24(2), 193\u2013202.","journal-title":"Journal of the Royal Statistical Society Series C: Applied Statistics"},{"key":"2840_CR23","unstructured":"Radford, A., Kim, J. W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., and others. (2021). Learning transferable visual models from natural language supervision. In International conference on machine learning, pages 8748\u20138763. PMLR."},{"key":"2840_CR24","unstructured":"Rafailov, R., Sharma, A., Mitchell, E., Manning, C. D., Ermon, S., & inn, C. (2024). Direct preference optimization: Your language model is secretly a reward model. Advances in Neural Information Processing Systems, 36."},{"key":"2840_CR25","doi-asserted-by":"crossref","unstructured":"Rohrbach, A., Hendricks, L. A., Burns, K., Darrell, T., & Saenko, K. (2018). Object hallucination in image captioning. arXiv preprint arXiv:1809.02156.","DOI":"10.18653\/v1\/D18-1437"},{"key":"2840_CR26","unstructured":"Schulman, J., Wolski, F., Dhariwal, P., Radford, A., & Klimov, O. (2017). Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347."},{"key":"2840_CR27","doi-asserted-by":"crossref","unstructured":"Shao, S., Li, Z., Zhang, T., Peng, C., Yu, G., Zhang, X., Li, J., & Sun, J. (2019). Objects365: A large-scale, high-quality dataset for object detection. In Proceedings of the IEEE\/CVF international conference on computer vision, pages 8430\u20138439.","DOI":"10.1109\/ICCV.2019.00852"},{"key":"2840_CR28","doi-asserted-by":"crossref","unstructured":"Sun, Z., Shen, S., Cao, S., Liu, H., Li, C., Shen, Y., Gan, C., Gui, L.-Y., Wang, Y.-X., Yang, Y., and others. (2024) Aligning large multimodal models with factually augmented rlhf. In Findings of the Association for Computational Linguistics.","DOI":"10.18653\/v1\/2024.findings-acl.775"},{"key":"2840_CR29","unstructured":"Suvorov, R., Logacheva, E., Mashikhin, A., Remizova, A., Ashukha, A., Silvestrov, A., Kong, N., Goka, H., Park, K., & Lempitsky, V. Resolution-robust large mask inpainting with fourier convolutions. In Proceedings of the IEEE\/CVF winter conference on applications of computer vision, pages 2149\u20132159."},{"key":"2840_CR30","first-page":"5824","volume":"33","author":"Yu Tianhe","year":"2020","unstructured":"Tianhe, Yu., Kumar, S., Gupta, A., Levine, S., Hausman, K., & Finn, C. (2020). Gradient surgery for multi-task learning. Advances in neural information processing systems, 33, 5824\u20135836.","journal-title":"Advances in neural information processing systems"},{"key":"2840_CR31","doi-asserted-by":"crossref","unstructured":"Tong, S., Liu, Z., Zhai, Y., Ma, Y., LeCun, Y., & Xie, S. (2024). Eyes wide shut? exploring the visual shortcomings of multimodal llms. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pages 9568\u20139578.","DOI":"10.1109\/CVPR52733.2024.00914"},{"key":"2840_CR32","unstructured":"Wang, J., Wang, Y., Xu, G., Zhang, J., Gu, Y., Jia, H., Yan, M., Zhang, J., & Sang, J. (2023). An llm-free multi-dimensional benchmark for mllms hallucination evaluation. arXiv preprint arXiv:2311.07397."},{"key":"2840_CR33","doi-asserted-by":"crossref","unstructured":"Wang, F., Zhou, W., Huang, J. Y., Xu, N., Zhang, S., Poon, H., & Chen, M. (2024). mdpo: Conditional preference optimization for multimodal large language models. arXiv preprint arXiv:2406.11839.","DOI":"10.18653\/v1\/2024.emnlp-main.460"},{"key":"2840_CR34","unstructured":"Wu, M., Ji, J., Huang, O., Li, J., Wu, Y., Sun, X., & Ji, R. (2024). Evaluating and analyzing relationship hallucinations in large vision-language models. In International Conference on Machine Learning, pages 53553\u201353570. PMLR."},{"key":"2840_CR35","doi-asserted-by":"crossref","unstructured":"Xie, Y., Li, G., Xu, X., & Kan, M. Y. (2024). V-dpo: Mitigating hallucination in large vision language models via vision-guided direct preference optimization. In Findings of the Association for Computational Linguistics: EMNLP, pages 13258\u201313273.","DOI":"10.18653\/v1\/2024.findings-emnlp.775"},{"key":"2840_CR36","unstructured":"Xing, Y., Li, Y., Laptev, I., & Lu, S. (2024). Mitigating object hallucination via concentric causal attention. Advances in neural information processing systems."},{"key":"2840_CR37","doi-asserted-by":"crossref","unstructured":"Yu, Q., Li, J., Wei, L., Pang, L., Ye, W., Qin, B., Tang, S., Tian, Q., & Zhuang, Y. (2024a) Hallucidoctor: Mitigating hallucinatory toxicity in visual instruction data. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pages 12944\u201312953.","DOI":"10.1109\/CVPR52733.2024.01230"},{"key":"2840_CR38","doi-asserted-by":"crossref","unstructured":"Yu, T., Yao, Y., Zhang, H., He, T., Han, Y., Cui, G., Hu, J., Liu, Z., Zheng, H.-T., Sun, M., and others. (2024a). Rlhf-v: Towards trustworthy mllms via behavior alignment from fine-grained correctional human feedback. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pages 13807\u201313816.","DOI":"10.1109\/CVPR52733.2024.01310"},{"key":"2840_CR39","doi-asserted-by":"crossref","unstructured":"Yu, T., Zhang, H., Li, Q., Xu, Q., Yao, Y., Chen, D., Lu, X., Cui, G., He, T., Liu, Z., Chua, T.-S., and others. (2024c) Rlaif-v: Aligning mllms through open-source ai feedback for super gpt-4v trustworthiness. arXiv preprint arXiv:2405.17220.","DOI":"10.1109\/CVPR52734.2025.01861"},{"key":"2840_CR40","doi-asserted-by":"crossref","unstructured":"Yue, Z., Zhang, L., & Jin, Q. (2024). Less is more: Mitigating multimodal hallucination from an eos decision perspective. In Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics, pages 11766\u201311781.","DOI":"10.18653\/v1\/2024.acl-long.633"},{"key":"2840_CR41","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Huang, X., Ma, J., Li, Z., Luo, Z., Xie, Y., Qin, Y., Luo, T., Li, Y., Liu, S., and others. (2023b). Recognize anything: A strong image tagging model. arXiv preprint arXiv:2306.03514.","DOI":"10.1109\/CVPRW63382.2024.00179"},{"key":"2840_CR42","unstructured":"Zhang, H., Li, F., Liu, S., Zhang, L., Su, H., Zhu, J., Ni, L., & Shum, H.-Y. (2023a). Dino: Detr with improved denoising anchor boxes for end-to-end object detection. In The Eleventh International Conference on Learning Representations. URL https:\/\/openreview.net\/forum?id=3mRwyG5one."},{"key":"2840_CR43","unstructured":"Zhang, M., Wu, W., Lu, Y., Song, Y., Rong, K., Yao, H., Zhao, J., Liu, F., Sun, Y., Feng, H., and others. (2024) Automated multi-level preference for mllms. Advances in Neural Information Processing Systems."},{"key":"2840_CR44","unstructured":"Zhang, Y. F., Yu, T., Tian, H., Fu, C., Li, P., Zeng, J., Xie, W., Shi, Y., Zhang, H., Wu, J., and others. (2025). Mm-rlhf: The next step forward in multimodal llm alignment. In International Conference on Machine Learning, pages 76625\u201376654. PMLR."},{"key":"2840_CR45","unstructured":"Zhao, L., Deng, Y., Zhang, W., & Gu, Q. (2024). Mitigating object hallucination in large vision-language models via classifier-free guidance. arXiv preprint arXiv:2402.08680."},{"key":"2840_CR46","unstructured":"Zhao, Z., Wang, B., Ouyang, L., Dong, X., Wang, J., & He, C. (2023) Beyond hallucinations: Enhancing lvlms through hallucination-aware direct preference optimization. arXiv preprint arXiv:2311.16839."},{"key":"2840_CR47","unstructured":"Zhou, Y., Cui, C., Rafailov, R., Finn, C., & Yao, H. (2024). Aligning modalities in vision large language models via preference fine-tuning. arXiv preprint arXiv:2402.11411."}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-026-02840-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-026-02840-0","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-026-02840-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,23]],"date-time":"2026-05-23T08:46:00Z","timestamp":1779525960000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-026-02840-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,13]]},"references-count":47,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2026,5]]}},"alternative-id":["2840"],"URL":"https:\/\/doi.org\/10.1007\/s11263-026-02840-0","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,4,13]]},"assertion":[{"value":"21 November 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"31 March 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 April 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflicts of Interest"}}],"article-number":"224"}}