{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,27]],"date-time":"2025-03-27T11:11:09Z","timestamp":1743073869201,"version":"3.40.3"},"publisher-location":"Singapore","reference-count":31,"publisher":"Springer Nature Singapore","isbn-type":[{"type":"print","value":"9789819784868"},{"type":"electronic","value":"9789819784875"}],"license":[{"start":{"date-parts":[[2024,11,4]],"date-time":"2024-11-04T00:00:00Z","timestamp":1730678400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,4]],"date-time":"2024-11-04T00:00:00Z","timestamp":1730678400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-97-8487-5_12","type":"book-chapter","created":{"date-parts":[[2024,11,3]],"date-time":"2024-11-03T07:02:46Z","timestamp":1730617366000},"page":"167-180","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Visual-Guided Reasoning Path Generation for Visual Question Answering"],"prefix":"10.1007","author":[{"given":"Xinyu","family":"Liu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chenchen","family":"Jing","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mingliang","family":"Zhai","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuwei","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yunde","family":"Jia","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,11,4]]},"reference":[{"key":"12_CR1","doi-asserted-by":"crossref","unstructured":"Anderson, P., He, X., Buehler, C., Teney, D., Johnson, M., Gould, S., Zhang, L.: Bottom-up and top-down attention for image captioning and visual question answering. In: Proceedings of the IEEE Conference on Computer Vision And Pattern Recognition, pp. 6077\u20136086 (2018)","DOI":"10.1109\/CVPR.2018.00636"},{"key":"12_CR2","doi-asserted-by":"crossref","unstructured":"Andreas, J., Rohrbach, M., Darrell, T., Klein, D.: Neural module networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 39\u201348 (2016)","DOI":"10.1109\/CVPR.2016.12"},{"key":"12_CR3","unstructured":"Asai, A., Hashimoto, K., Hajishirzi, H., Socher, R., Xiong, C.: Learning to retrieve reasoning paths over wikipedia graph for question answering (2019). arXiv:1911.10470"},{"key":"12_CR4","first-page":"1877","volume":"33","author":"T Brown","year":"2020","unstructured":"Brown, T., Mann, B., Ryder, N., Subbiah, M., Kaplan, J.D., Dhariwal, P., Neelakantan, A., Shyam, P., Sastry, G., Askell, A., et al.: Language models are few-shot learners. Adv. Neural. Inf. Process. Syst. 33, 1877\u20131901 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"12_CR5","doi-asserted-by":"crossref","unstructured":"Cao, Q., Liang, X., Wang, K., Lin, L.: Linguistically driven graph capsule network for visual question reasoning (2020). arXiv:2003.10065","DOI":"10.1109\/ICCV48922.2021.00164"},{"key":"12_CR6","doi-asserted-by":"crossref","unstructured":"Chen, W., Gan, Z., Li, L., Cheng, Y., Wang, W., Liu, J.: Meta module network for compositional visual reasoning. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 655\u2013664 (2021)","DOI":"10.1109\/WACV48630.2021.00070"},{"key":"12_CR7","doi-asserted-by":"crossref","unstructured":"Do, T., Do, T.T., Tran, H., Tjiputra, E., Tran, Q.D.: Compact trilinear interaction for visual question answering. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 392\u2013401 (2019)","DOI":"10.1109\/ICCV.2019.00048"},{"key":"12_CR8","doi-asserted-by":"crossref","unstructured":"Goyal, Y., Khot, T., Summers-Stay, D., Batra, D., Parikh, D.: Making the v in vqa matter: elevating the role of image understanding in visual question answering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6904\u20136913 (2017)","DOI":"10.1109\/CVPR.2017.670"},{"key":"12_CR9","unstructured":"Guo, D., Xu, C., Tao, D.: Graph reasoning networks for visual question answering (2019). arXiv:1907.09815"},{"key":"12_CR10","doi-asserted-by":"crossref","unstructured":"Gupta, T., Kembhavi, A.: Visual programming: Compositional visual reasoning without training. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14953\u201314962 (2023)","DOI":"10.1109\/CVPR52729.2023.01436"},{"key":"12_CR11","doi-asserted-by":"crossref","unstructured":"Hu, R., Andreas, J., Darrell, T., Saenko, K.: Explainable neural computation via stack neural module networks. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 53\u201369 (2018)","DOI":"10.1007\/978-3-030-01234-2_4"},{"key":"12_CR12","doi-asserted-by":"crossref","unstructured":"Hu, R., Rohrbach, A., Darrell, T., Saenko, K.: Language-conditioned graph networks for relational reasoning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 10294\u201310303 (2019)","DOI":"10.1109\/ICCV.2019.01039"},{"key":"12_CR13","doi-asserted-by":"publisher","unstructured":"Hudson, D.A., Manning, C.D.: Compositional Attention Networks for Machine Reasoning (2018). https:\/\/doi.org\/10.48550\/arXiv.1803.03067","DOI":"10.48550\/arXiv.1803.03067"},{"key":"12_CR14","doi-asserted-by":"crossref","unstructured":"Hudson, D.A., Manning, C.D.: GQA: a new dataset for real-world visual reasoning and compositional question answering. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6700\u20136709 (2019)","DOI":"10.1109\/CVPR.2019.00686"},{"key":"12_CR15","doi-asserted-by":"crossref","unstructured":"Li, G., Wang, X., Zhu, W.: Perceptual visual reasoning with knowledge propagation. In: Proceedings of the 27th ACM International Conference on Multimedia, pp. 530\u2013538 (2019)","DOI":"10.1145\/3343031.3350922"},{"key":"12_CR16","unstructured":"Li, J., Li, D., Savarese, S., Hoi, S.: Blip-2: bootstrapping language-image pre-training with frozen image encoders and large language models (2023). arXiv:2301.12597"},{"key":"12_CR17","unstructured":"Liang, W., Niu, F., Reganti, A., Thattai, G., Tur, G.: Lrta: a transparent neural-symbolic reasoning framework with modular supervision for visual question answering (2020). arXiv:2011.10731"},{"key":"12_CR18","unstructured":"Liu, H., Li, C., Wu, Q., Lee, Y.J.: Visual instruction tuning (2023). arXiv:2304.08485"},{"key":"12_CR19","unstructured":"Mao, J., Gan, C., Kohli, P., Tenenbaum, J.B., Wu, J.: The neuro-symbolic concept learner: interpreting scenes, words, and sentences from natural supervision (2019). arXiv:1904.12584"},{"key":"12_CR20","doi-asserted-by":"crossref","unstructured":"Pennington, J., Socher, R., Manning, C.: GloVe: global vectors for word representation. In: Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing (EMNLP), pp. 1532\u20131543. Association for Computational Linguistics, Doha, Qatar (2014). https:\/\/aclanthology.org\/D14-1162","DOI":"10.3115\/v1\/D14-1162"},{"key":"12_CR21","doi-asserted-by":"crossref","unstructured":"Tan, H., Bansal, M.: Lxmert: learning cross-modality encoder representations from transformers (2019). arXiv:1908.07490","DOI":"10.18653\/v1\/D19-1514"},{"key":"12_CR22","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, \u0141., Polosukhin, I.: Attention is all you need. Adv. Neural Inf. Process. Syst. 30 (2017)"},{"key":"12_CR23","doi-asserted-by":"publisher","unstructured":"Wang, X., Wei, J., Schuurmans, D., Le, Q., Chi, E., Narang, S., Chowdhery, A., Zhou, D.: Self-Consistency Improves Chain of Thought Reasoning in Language Models (2022). https:\/\/doi.org\/10.48550\/arXiv.2203.11171","DOI":"10.48550\/arXiv.2203.11171"},{"key":"12_CR24","unstructured":"Wei, J., Tay, Y., Bommasani, R., Raffel, C., Zoph, B., Borgeaud, S., Yogatama, D., Bosma, M., Zhou, D., Metzler, D., et\u00a0al.: Emergent abilities of large language models (2022). arXiv:2206.07682"},{"key":"12_CR25","first-page":"24824","volume":"35","author":"J Wei","year":"2022","unstructured":"Wei, J., Wang, X., Schuurmans, D., Bosma, M., Xia, F., Chi, E., Le, Q.V., Zhou, D., et al.: Chain-of-thought prompting elicits reasoning in large language models. Adv. Neural. Inf. Process. Syst. 35, 24824\u201324837 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"12_CR26","doi-asserted-by":"publisher","first-page":"229","DOI":"10.1007\/BF00992696","volume":"8","author":"RJ Williams","year":"1992","unstructured":"Williams, R.J.: Simple statistical gradient-following algorithms for connectionist reinforcement learning. Mach. Learn. 8, 229\u2013256 (1992)","journal-title":"Mach. Learn."},{"key":"12_CR27","doi-asserted-by":"crossref","unstructured":"Xu, W., Deng, Y., Zhang, H., Cai, D., Lam, W.: Exploiting reasoning chains for multi-hop science question answering (2021). arXiv:2109.02905","DOI":"10.18653\/v1\/2021.findings-emnlp.99"},{"key":"12_CR28","unstructured":"Ye, Q., Xu, H., Xu, G., Ye, J., Yan, M., Zhou, Y., Wang, J., Hu, A., Shi, P., Shi, Y., et\u00a0al.: mplug-owl: modularization empowers large language models with multimodality (2023). arXiv:2304.14178"},{"key":"12_CR29","doi-asserted-by":"crossref","unstructured":"Yu, Z., Yu, J., Cui, Y., Tao, D., Tian, Q.: Deep modular co-attention networks for visual question answering. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6281\u20136290 (2019)","DOI":"10.1109\/CVPR.2019.00644"},{"key":"12_CR30","unstructured":"Zelikman, E., Wu, Y., Mu, J., Goodman, N.D.: Star: Bootstrapping reasoning with reasoning (2022). arXiv:2203.14465"},{"key":"12_CR31","first-page":"17021","volume":"34","author":"Z Zhao","year":"2021","unstructured":"Zhao, Z., Samel, K., Chen, B., et al.: Proto: program-guided transformer for program-guided tasks. Adv. Neural. Inf. Process. Syst. 34, 17021\u201317036 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."}],"container-title":["Lecture Notes in Computer Science","Pattern Recognition and Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-97-8487-5_12","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,3]],"date-time":"2024-11-03T07:06:28Z","timestamp":1730617588000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-97-8487-5_12"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,4]]},"ISBN":["9789819784868","9789819784875"],"references-count":31,"URL":"https:\/\/doi.org\/10.1007\/978-981-97-8487-5_12","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2024,11,4]]},"assertion":[{"value":"4 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"PRCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Chinese Conference on Pattern Recognition and Computer Vision  (PRCV)","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Urumqi","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18 October 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"20 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"7","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ccprcv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/2024.prcv.cn\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}