{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,22]],"date-time":"2026-07-22T23:35:32Z","timestamp":1784763332794,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":61,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,10,26]],"date-time":"2023-10-26T00:00:00Z","timestamp":1698278400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"name":"Natural Science Foundation of China","award":["62206097"],"award-info":[{"award-number":["62206097"]}]},{"name":"Shanghai Pujiang Talent Program","award":["22PJ1403000"],"award-info":[{"award-number":["22PJ1403000"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,10,26]]},"DOI":"10.1145\/3581783.3612389","type":"proceedings-article","created":{"date-parts":[[2023,10,27]],"date-time":"2023-10-27T07:26:54Z","timestamp":1698391614000},"page":"4389-4400","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":28,"title":["Improving Zero-shot Visual Question Answering via Large Language Models with Reasoning Question Prompts"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0192-8498","authenticated-orcid":false,"given":"Yunshi","family":"Lan","sequence":"first","affiliation":[{"name":"East China Normal University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-8818-0269","authenticated-orcid":false,"given":"Xiang","family":"Li","sequence":"additional","affiliation":[{"name":"East China Normal University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-0341-3860","authenticated-orcid":false,"given":"Xin","family":"Liu","sequence":"additional","affiliation":[{"name":"East China Normal University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4155-2859","authenticated-orcid":false,"given":"Yang","family":"Li","sequence":"additional","affiliation":[{"name":"Alibaba Group, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9513-4372","authenticated-orcid":false,"given":"Wei","family":"Qin","sequence":"additional","affiliation":[{"name":"Hefei University of Technology, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4132-8630","authenticated-orcid":false,"given":"Weining","family":"Qian","sequence":"additional","affiliation":[{"name":"East China Normal University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2023,10,27]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00522"},{"key":"e_1_3_2_2_2_1","unstructured":"Jean-Baptiste Alayrac et al. 2022. Flamingo: a visual language model for fewshot learning. In Advances in Neural Information Processing Systems 23716--23736."},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.279"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"crossref","unstructured":"Pratyay Banerjee Tejas Gokhale Yezhou Yang and Chitta Baral. 2020. Weaqa: weak supervision via captions for visual question answering. arXiv preprint arXiv:2012.02356.","DOI":"10.18653\/v1\/2021.findings-acl.302"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"crossref","unstructured":"Sid Black et al. 2022. Gpt-neox-20b: an open-source autoregressive language model. arXiv: 2204.06745 [cs.CL].","DOI":"10.18653\/v1\/2022.bigscience-1.9"},{"key":"e_1_3_2_2_6_1","unstructured":"Tom Brown et al. 2020. Language models are few-shot learners. Advances in neural information processing systems 1877--1901."},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"crossref","unstructured":"Soravit Changpinyo Doron Kukliansky Idan Szpektor Xi Chen Nan Ding and Radu Soricut. 2022. All you may need for vqa are image captions. arXiv preprint arXiv:2205.01883.","DOI":"10.18653\/v1\/2022.naacl-main.142"},{"key":"e_1_3_2_2_8_1","unstructured":"Zhenfang Chen Qinhong Zhou Yikang Shen Yining Hong Hao Zhang and Chuang Gan. 2023. See think confirm: interactive prompting between vision and language models for knowledge-based visual reasoning. arXiv preprint arXiv:2301.05226."},{"key":"e_1_3_2_2_9_1","volume-title":"Proceedings of the 38th International Conference on Machine Learning","author":"Cho Jaemin","year":"2021","unstructured":"Jaemin Cho, Jie Lei, Hao Tan, and Mohit Bansal. 2021. Unifying vision-andlanguage tasks via text generation. In Proceedings of the 38th International Conference on Machine Learning, 1931--1942."},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.findings-acl.187"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00501"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3487553.3524648"},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.findings-emnlp.44"},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.670"},{"key":"e_1_3_2_2_15_1","unstructured":"Liangke Gui Borui Wang Qiuyuan Huang Alex Hauptmann Yonatan Bisk and Jianfeng Gao. 2021. Kat: a knowledge augmented transformer for visionand- language. arXiv preprint arXiv:2112.08614."},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01046"},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"crossref","unstructured":"Jingjing Jiang and Nanning Zheng. 2023. Mixphm: redundancy-aware parameterefficient tuning for low-resource visual question answering. arXiv preprint arXiv:2303.01239.","DOI":"10.1109\/CVPR52729.2023.02318"},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.197"},{"key":"e_1_3_2_2_19_1","volume-title":"Sentence-level fluency evaluation: references help, but can be spared! In Proceedings of the 22nd Conference on Computational Natural Language Learning, 313--323","author":"Kann Katharina","unstructured":"Katharina Kann, Sascha Rothe, and Katja Filippova. 2018. Sentence-level fluency evaluation: references help, but can be spared! In Proceedings of the 22nd Conference on Computational Natural Language Learning, 313--323."},{"key":"e_1_3_2_2_20_1","volume-title":"Hyunsoo Cho, Hwiyeol Jo, Sang-Woo Lee, Sang-goo Lee, Kang Min Yoo, and Taeuk Kim.","author":"Kim Junyeob","year":"2022","unstructured":"Junyeob Kim, Hyuhng Joon Kim, Hyunsoo Cho, Hwiyeol Jo, Sang-Woo Lee, Sang-goo Lee, Kang Min Yoo, and Taeuk Kim. 2022. Ground-truth labels matter: a deeper look into input-label demonstrations. arXiv preprint arXiv:2205.12685."},{"key":"e_1_3_2_2_21_1","unstructured":"Hannah Rose Kirk Yennie Jun Filippo Volpin Haider Iqbal Elias Benussi Frederic Dreyer Aleksandar Shtedritski and Yuki Asano. 2021. Bias out-ofthe-box: an empirical analysis of intersectional occupational biases in popular generative language models. Advances in neural information processing systems 2611--2624."},{"key":"e_1_3_2_2_22_1","volume-title":"Machel Reid, Yutaka Matsuo, and Yusuke Iwasawa.","author":"Kojima Takeshi","year":"2022","unstructured":"Takeshi Kojima, Shixiang Shane Gu, Machel Reid, Yutaka Matsuo, and Yusuke Iwasawa. 2022. Large language models are zero-shot reasoners. arXiv preprint arXiv:2205.11916."},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.707"},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i10.21346"},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.acl-long.353"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"crossref","unstructured":"Sheng Liang Mengjie Zhao and Hinrich Sch\u00fctze. 2022. Modular and parameterefficient multimodal fusion with prompting. arXiv preprint arXiv:2203.08055.","DOI":"10.18653\/v1\/2022.findings-acl.234"},{"key":"e_1_3_2_2_27_1","volume-title":"Computer Vision--ECCV 2014: 13th European Conference, 740--755.","author":"Lin Tsung-Yi","unstructured":"Tsung-Yi Lin, Michael Maire, Serge Belongie, James Hays, Pietro Perona, Deva Ramanan, Piotr Doll\u00e1r, and C Lawrence Zitnick. 2014. Microsoft coco: common objects in context. In Computer Vision--ECCV 2014: 13th European Conference, 740--755."},{"key":"e_1_3_2_2_28_1","unstructured":"Yuanze Lin Yujia Xie Dongdong Chen Yichong Xu Chenguang Zhu and Lu Yuan. 2022. Revive: regional visual representation matters in knowledge-based visual question answering. arXiv preprint arXiv:2206.01201."},{"key":"e_1_3_2_2_29_1","unstructured":"Jiachang Liu Dinghan Shen Yizhe Zhang Bill Dolan Lawrence Carin and Weizhu Chen. 2021. What makes good in-context examples for gpt-3? arXiv preprint arXiv:2101.06804."},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"crossref","unstructured":"Oscar Ma\u00f1as Pau Rodriguez Saba Ahmadi Aida Nematzadeh Yash Goyal and Aishwarya Agrawal. 2022. Mapl: parameter-efficient adaptation of unimodal pre-trained models for vision-language few-shot prompting. arXiv preprint arXiv:2210.07179.","DOI":"10.18653\/v1\/2023.eacl-main.185"},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01389"},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00331"},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"crossref","unstructured":"Ning Miao Hao Zhou Lili Mou Rui Yan and Lei Li. 2019. Cgmh: constrained sentence generation by metropolis-hastings sampling. In Proceedings of the Thirty-Third AAAI Conference on Artificial Intelligence and Thirty-First Innovative Applications of Artificial Intelligence Conference and Ninth AAAI Symposium on Educational Advances in Artificial Intelligence 6834--6842.","DOI":"10.1609\/aaai.v33i01.33016834"},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"crossref","unstructured":"Sewon Min Xinxi Lyu Ari Holtzman Mikel Artetxe Mike Lewis Hannaneh Hajishirzi and Luke Zettlemoyer. 2022. Rethinking the role of demonstrations: what makes in-context learning work? arXiv preprint arXiv:2202.12837.","DOI":"10.18653\/v1\/2022.emnlp-main.759"},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.244"},{"key":"e_1_3_2_2_36_1","unstructured":"Medhini Narasimhan Svetlana Lazebnik and Alexander Schwing. 2018. Out of the box: reasoning with graph convolution nets for factual visual question answering. In Advances in Neural Information Processing Systems."},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.naacl-main.410"},{"key":"e_1_3_2_2_38_1","volume-title":"International conference on machine learning, 8748--8763","author":"Alec","unstructured":"Alec Radford et al. 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning, 8748--8763."},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"crossref","unstructured":"Teven Le Scao et al. 2022. What language model to train if you have one million gpu hours? arXiv preprint arXiv: 2210.15424.","DOI":"10.18653\/v1\/2022.findings-emnlp.54"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"crossref","unstructured":"Timo Schick and Hinrich Sch\u00fctze. 2020. Exploiting cloze questions for few shot text classification and natural language inference. arXiv preprint arXiv:2001.07676.","DOI":"10.18653\/v1\/2021.eacl-main.20"},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"crossref","unstructured":"Patrick Schramowski Cigdem Turan Nico Andersen Constantin A. Rothkopf and Kristian Kersting. 2022. Large pre-trained language models contain humanlike biases of what is right and wrong to do. Nature Machine Intelligence 258--268.","DOI":"10.1038\/s42256-022-00458-8"},{"key":"e_1_3_2_2_42_1","volume-title":"Tel Aviv","author":"Schwenk Dustin","year":"2022","unstructured":"Dustin Schwenk, Apoorv Khandelwal, Christopher Clark, Kenneth Marino, and Roozbeh Mottaghi. 2022. A-okvqa: a benchmark for visual question answering using world knowledge. In Computer Vision-ECCV 2022: 17th European Conference, Tel Aviv, Israel, October 23-27, 2022, Proceedings, Part VIII, 146--162."},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"crossref","unstructured":"Zhenwei Shao Zhou Yu Meng Wang and Jun Yu. 2023. Prompting large language models with answer heuristics for knowledge-based visual question answering. arXiv preprint arXiv:2303.01903.","DOI":"10.1109\/CVPR52729.2023.01438"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.421"},{"key":"e_1_3_2_2_45_1","unstructured":"Maria Tsimpoukelli Jacob Menick Serkan Cabi S. M. Ali Eslami Oriol Vinyals and Felix Hill. 2021. Multimodal few-shot learning with frozen language models. In Advances in Neural Information Processing Systems 200--212."},{"key":"e_1_3_2_2_46_1","unstructured":"Maria Tsimpoukelli Jacob L Menick Serkan Cabi SM Eslami Oriol Vinyals and Felix Hill. 2021. Multimodal few-shot learning with frozen language models. Advances in Neural Information Processing Systems 200--212."},{"key":"e_1_3_2_2_47_1","unstructured":"Ben Wang and Aran Komatsuzaki. 2021. Gpt-j-6b: a 6 billion parameter autoregressive language model. github."},{"key":"e_1_3_2_2_48_1","unstructured":"PengWang QiWu Chunhua Shen Anton van den Hengel and Anthony Dick. 2015. Explicit knowledge-based reasoning for visual question answering. arXiv preprint arXiv:1511.02570."},{"key":"e_1_3_2_2_49_1","volume-title":"Chi, and Denny Zhou","author":"Wang Xuezhi","year":"2022","unstructured":"Xuezhi Wang, Jason Wei, Dale Schuurmans, Quoc Le, Ed Chi, and Denny Zhou. 2022. Self-consistency improves chain of thought reasoning in language models. arXiv preprint arXiv:2203.11171."},{"key":"e_1_3_2_2_50_1","volume-title":"Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing, 5085--5109","author":"Yizhong","unstructured":"Yizhong Wang et al. 2022. Super-naturalinstructions:generalization via declarative instructions on 1600 tasks. In Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing, 5085--5109."},{"key":"e_1_3_2_2_51_1","volume-title":"Brian Lester, Nan Du, Andrew M Dai, and Quoc V Le.","author":"Wei Jason","year":"2021","unstructured":"Jason Wei, Maarten Bosma, Vincent Y Zhao, Kelvin Guu, Adams Wei Yu, Brian Lester, Nan Du, Andrew M Dai, and Quoc V Le. 2021. Finetuned language models are zero-shot learners. arXiv preprint arXiv:2109.01652."},{"key":"e_1_3_2_2_52_1","volume-title":"Chi, Quoc Le, and Denny Zhou","author":"Wei Jason","year":"2022","unstructured":"Jason Wei, Xuezhi Wang, Dale Schuurmans, Maarten Bosma, Ed Chi, Quoc Le, and Denny Zhou. 2022. Chain of thought prompting elicits reasoning in large language models. arXiv preprint arXiv:2201.11903."},{"key":"e_1_3_2_2_53_1","volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence, 2712--2721","author":"Wu Jialin","year":"2022","unstructured":"Jialin Wu, Jiasen Lu, Ashish Sabharwal, and Roozbeh Mottaghi. 2022. Multimodal answer validation for knowledge-based vqa. In Proceedings of the AAAI Conference on Artificial Intelligence, 2712--2721."},{"key":"e_1_3_2_2_54_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i3.20215"},{"key":"e_1_3_2_2_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00553"},{"key":"e_1_3_2_2_56_1","unstructured":"Susan Zhang et al. 2022. Opt: open pre-trained transformer language models. arXiv preprint arXiv:2205.01068."},{"key":"e_1_3_2_2_57_1","unstructured":"Zhuosheng Zhang Aston Zhang Mu Li and Alex Smola. 2022. Automatic chain of thought prompting in large language models. arXiv preprint arXiv:2210.03493."},{"key":"e_1_3_2_2_58_1","volume-title":"International Conference on Machine Learning, 12697--12706","author":"Zhao Zihao","year":"2021","unstructured":"Zihao Zhao, Eric Wallace, Shi Feng, Dan Klein, and Sameer Singh. 2021. Calibrate before use: improving few-shot performance of language models. In International Conference on Machine Learning, 12697--12706."},{"key":"e_1_3_2_2_59_1","unstructured":"Denny Zhou et al. 2022. Least-to-most prompting enables complex reasoning in large language models. arXiv preprint arXiv:2205.10625."},{"key":"e_1_3_2_2_60_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2020\/153"},{"key":"e_1_3_2_2_61_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.coling-main.169"}],"event":{"name":"MM '23: The 31st ACM International Conference on Multimedia","location":"Ottawa ON Canada","acronym":"MM '23","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 31st ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612389","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3581783.3612389","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T23:55:28Z","timestamp":1755820528000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612389"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,26]]},"references-count":61,"alternative-id":["10.1145\/3581783.3612389","10.1145\/3581783"],"URL":"https:\/\/doi.org\/10.1145\/3581783.3612389","relation":{},"subject":[],"published":{"date-parts":[[2023,10,26]]},"assertion":[{"value":"2023-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}