{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T15:36:30Z","timestamp":1783438590013,"version":"3.54.6"},"publisher-location":"Cham","reference-count":57,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031726729","type":"print"},{"value":"9783031726736","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,10,22]],"date-time":"2024-10-22T00:00:00Z","timestamp":1729555200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,22]],"date-time":"2024-10-22T00:00:00Z","timestamp":1729555200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-72673-6_9","type":"book-chapter","created":{"date-parts":[[2024,10,21]],"date-time":"2024-10-21T16:03:50Z","timestamp":1729526630000},"page":"156-172","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["SQ-LLaVA: Self-Questioning for\u00a0Large Vision-Language Assistant"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-0935-6196","authenticated-orcid":false,"given":"Guohao","family":"Sun","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0712-5378","authenticated-orcid":false,"given":"Can","family":"Qin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0074-0274","authenticated-orcid":false,"given":"Jiamian","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-2471-5449","authenticated-orcid":false,"given":"Zeyuan","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-4585-5261","authenticated-orcid":false,"given":"Ran","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5639-7540","authenticated-orcid":false,"given":"Zhiqiang","family":"Tao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,10,22]]},"reference":[{"key":"9_CR1","unstructured":"Gpt-4v(ision) system card (2023)"},{"key":"9_CR2","doi-asserted-by":"crossref","unstructured":"Agrawal, H., et al.: Nocaps: novel object captioning at scale. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00904"},{"key":"9_CR3","unstructured":"Bai, J., et al.: Qwen-vl: a versatile vision-language model for understanding, localization, text reading, and beyond. arXiv (2023)"},{"key":"9_CR4","unstructured":"Brown, T., et al.: Language models are few-shot learners. In: NeurIPS (2020)"},{"key":"9_CR5","doi-asserted-by":"crossref","unstructured":"Chen, L., et al.: Sharegpt4v: improving large multi-modal models with better captions. arXiv (2023)","DOI":"10.1007\/978-3-031-72643-9_22"},{"key":"9_CR6","unstructured":"Chen, X., et al.: Microsoft coco captions: data collection and evaluation server. arXiv (2015)"},{"key":"9_CR7","unstructured":"Chowdhery, A., et al.: Palm: scaling language modeling with pathways. J. Mach. Learn. Res. (2022)"},{"key":"9_CR8","unstructured":"Dai, W., et al.: Instructblip: towards general-purpose vision-language models with instruction tuning. In: NeurIPS (2023)"},{"key":"9_CR9","doi-asserted-by":"crossref","unstructured":"Dess\u00ec, R., Bevilacqua, M., Gualdoni, E., Rakotonirina, N.C., Franzon, F., Baroni, M.: Cross-domain image captioning with discriminative finetuning. In: CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.00670"},{"key":"9_CR10","doi-asserted-by":"crossref","unstructured":"Goyal, Y., Khot, T., Summers-Stay, D., Batra, D., Parikh, D.: Making the V in VQA matter: elevating the role of image understanding in visual question answering. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.670"},{"key":"9_CR11","doi-asserted-by":"crossref","unstructured":"Gurari, D., et al.: Vizwiz grand challenge: answering visual questions from blind people. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00380"},{"key":"9_CR12","unstructured":"Hu, E.J., et al.: Lora: low-rank adaptation of large language models. In: ICLR (2022)"},{"key":"9_CR13","doi-asserted-by":"crossref","unstructured":"Hudson, D.A., Manning, C.D.: GQA: a new dataset for real-world visual reasoning and compositional question answering. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00686"},{"key":"9_CR14","doi-asserted-by":"crossref","unstructured":"Ikotun, A.M., Ezugwu, A.E., Abualigah, L., Abuhaija, B., Heming, J.: K-means clustering algorithms: a comprehensive review, variants analysis, and advances in the era of big data. Inf. Sci. (2023)","DOI":"10.1016\/j.ins.2022.11.139"},{"key":"9_CR15","doi-asserted-by":"crossref","unstructured":"Kazemzadeh, S., Ordonez, V., Matten, M., Berg, T.: ReferItGame: referring to objects in photographs of natural scenes. In: EMNLP (2014)","DOI":"10.3115\/v1\/D14-1086"},{"key":"9_CR16","doi-asserted-by":"crossref","unstructured":"Kirillov, A., et al.: Segment anything. In: ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"9_CR17","unstructured":"Krishna, R., et al.: Visual genome: connecting language and vision using crowdsourced dense image annotations. In: IJCV (2016)"},{"key":"9_CR18","unstructured":"Li, J., et al.: Empowering vision-language models to follow interleaved vision-language instructions. In: ICLR (2024)"},{"key":"9_CR19","unstructured":"Li, J., Li, D., Xiong, C., Hoi, S.C.H.: Blip: bootstrapping language-image pre-training for unified vision-language understanding and generation. In: International Conference on Machine Learning (2022)"},{"key":"9_CR20","doi-asserted-by":"crossref","unstructured":"Li, Y., Du, Y., Zhou, K., Wang, J., Zhao, X., Wen, J.R.: Evaluating object hallucination in large vision-language models. In: EMNLP (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.20"},{"key":"9_CR21","doi-asserted-by":"crossref","unstructured":"Liu, H., Li, C., Li, Y., Lee, Y.J.: Improved baselines with visual instruction tuning. arXiv (2023)","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"9_CR22","unstructured":"Liu, H., Li, C., Wu, Q., Lee, Y.J.: Visual instruction tuning. In: NeurIPS (2023)"},{"key":"9_CR23","doi-asserted-by":"crossref","unstructured":"Liu, Y., et al.: Mmbench: is your multi-modal model an all-around player? arXiv (2023)","DOI":"10.1007\/978-3-031-72658-3_13"},{"key":"9_CR24","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. In: ICLR (2019)"},{"key":"9_CR25","unstructured":"Lu, P., et al.: Learn to explain: multimodal reasoning via thought chains for science question answering. In: NeurIPS (2022)"},{"key":"9_CR26","doi-asserted-by":"crossref","unstructured":"Ma\u00f1as, O., L\u00f3pez, P.R., Ahmadi, S., Nematzadeh, A., Goyal, Y., Agrawal, A.: MAPL: parameter-efficient adaptation of unimodal pre-trained models for vision-language few-shot prompting. In: EACL (2023)","DOI":"10.18653\/v1\/2023.eacl-main.185"},{"key":"9_CR27","doi-asserted-by":"crossref","unstructured":"Mishra, A., Shekhar, S., Singh, A.K., Chakraborty, A.: OCR-VQA: visual question answering by reading text in images. In: ICDAR (2019)","DOI":"10.1109\/ICDAR.2019.00156"},{"key":"9_CR28","unstructured":"Mokady, R.: Clipcap: clip prefix for image captioning. arXiv (2021)"},{"key":"9_CR29","unstructured":"Ordonez, V., Kulkarni, G., Berg, T.: Im2text: describing images using 1 million captioned photographs. In: NeurIPS (2011)"},{"key":"9_CR30","unstructured":"Ouyang, L., et al.: Training language models to follow instructions with human feedback. In: NeurIPS (2022)"},{"key":"9_CR31","doi-asserted-by":"crossref","unstructured":"Papineni, K., Roukos, S., Ward, T., Zhu, W.J.: Bleu: a method for automatic evaluation of machine translation. In: ACL (2002)","DOI":"10.3115\/1073083.1073135"},{"key":"9_CR32","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: ICML (2021)"},{"key":"9_CR33","doi-asserted-by":"crossref","unstructured":"Rohrbach, A., Hendricks, L.A., Burns, K., Darrell, T., Saenko, K.: Object hallucination in image captioning. In: EMNLP (2018)","DOI":"10.18653\/v1\/D18-1437"},{"key":"9_CR34","unstructured":"Sanh, V., et al.: Multitask prompted training enables zero-shot task generalization. In: ICLR (2022)"},{"key":"9_CR35","unstructured":"Schuhmann, C., et al.: Laion-400m: open dataset of clip-filtered 400 million image-text pairs. In: NeurIPS Workshop (2021)"},{"key":"9_CR36","doi-asserted-by":"crossref","unstructured":"Sharma, P., Ding, N., Goodman, S., Soricut, R.: Conceptual captions: a cleaned, hypernymed, image alt-text dataset for automatic image captioning. In: Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers) (2018)","DOI":"10.18653\/v1\/P18-1238"},{"key":"9_CR37","doi-asserted-by":"crossref","unstructured":"Shukor, M., Dancette, C., Cord, M.: eP-ALM: efficient perceptual augmentation of language models. In: ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.02016"},{"key":"9_CR38","doi-asserted-by":"crossref","unstructured":"Singh, A., et al.: Towards VQA models that can read. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00851"},{"key":"9_CR39","doi-asserted-by":"crossref","unstructured":"Sun, G., Bai, Y., Yang, X., Fang, Y., Fu, Y., Tao, Z.: Aligning out-of-distribution web images and caption semantics via evidential learning. In: Proceedings of the ACM on Web Conference 2024 (2024)","DOI":"10.1145\/3589334.3645653"},{"key":"9_CR40","doi-asserted-by":"crossref","unstructured":"Tofade, T., Elsner, J., Haines, S.: Best practice strategies for effective use of questions as a teaching tool. Am. J. Pharm. Educ. (2013)","DOI":"10.5688\/ajpe777155"},{"key":"9_CR41","unstructured":"Touvron, H., et al.: Llama: open and efficient foundation language models. arXiv (2023)"},{"key":"9_CR42","unstructured":"Touvron, H., et al.: Llama 2: open foundation and fine-tuned chat models. arXiv (2023)"},{"key":"9_CR43","unstructured":"Triantafillou, E., et al.: Meta-dataset: a dataset of datasets for learning to learn from few examples. In: ICLR (2019)"},{"key":"9_CR44","doi-asserted-by":"crossref","unstructured":"Vattani, A.: K-means requires exponentially many iterations even in the plane. In: Annual Symposium on Computational Geometry (2009)","DOI":"10.1145\/1542362.1542419"},{"key":"9_CR45","doi-asserted-by":"crossref","unstructured":"Vedantam, R., Zitnick, C.L., Parikh, D.: Cider: consensus-based image description evaluation. In: CVPR (2015)","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"9_CR46","doi-asserted-by":"crossref","unstructured":"Wang, J., et al.: Text is mass: modeling as stochastic embedding for text-video retrieval. In: CVPR (2024)","DOI":"10.1109\/CVPR52733.2024.01566"},{"key":"9_CR47","doi-asserted-by":"crossref","unstructured":"Wang, Y., et al.: Self-instruct: aligning language models with self-generated instructions. In: ACL (2022)","DOI":"10.18653\/v1\/2023.acl-long.754"},{"key":"9_CR48","doi-asserted-by":"crossref","unstructured":"Wang, Y., et al.: Super-naturalinstructions: generalization via declarative instructions on 1600+ NLP tasks. In: EMNLP (2022)","DOI":"10.18653\/v1\/2022.emnlp-main.340"},{"key":"9_CR49","unstructured":"Wei, J., et al.: Finetuned language models are zero-shot learners. In: ICLR (2022)"},{"key":"9_CR50","unstructured":"Xu, C., et al.: Wizardlm: empowering large language models to follow complex instructions. In: ICLR (2024)"},{"key":"9_CR51","doi-asserted-by":"crossref","unstructured":"Young, P., Lai, A., Hodosh, M., Hockenmaier, J.: From image descriptions to visual denotations: new similarity metrics for semantic inference over event descriptions. In: TACL (2014)","DOI":"10.1162\/tacl_a_00166"},{"key":"9_CR52","unstructured":"Yu, W., et al.: Mm-vet: evaluating large multimodal models for integrated capabilities. arXiv (2023)"},{"key":"9_CR53","unstructured":"Zhang, R., et al.: Llama-adapter: efficient fine-tuning of language models with zero-init attention. In: ICLR (2024)"},{"key":"9_CR54","unstructured":"Zhang, Y., et al.: Llavar: enhanced visual instruction tuning for text-rich image understanding. arXiv (2023)"},{"key":"9_CR55","unstructured":"Zhao, B., Wu, B., Huang, T.: Svit: scaling up visual instruction tuning. arXiv (2023)"},{"key":"9_CR56","unstructured":"Zheng, L., et al.: Judging LLM-as-a-judge with MT-bench and chatbot arena. In: NeurIPS (2023)"},{"key":"9_CR57","unstructured":"Zhu, D., Chen, J., Shen, X., Li, X., Elhoseiny, M.: Minigpt-4: enhancing vision-language understanding with advanced large language models. In: ICLR (2024)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72673-6_9","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,30]],"date-time":"2024-11-30T00:03:40Z","timestamp":1732925020000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72673-6_9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,22]]},"ISBN":["9783031726729","9783031726736"],"references-count":57,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72673-6_9","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,22]]},"assertion":[{"value":"22 October 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}