{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,26]],"date-time":"2025-03-26T12:28:15Z","timestamp":1742992095492,"version":"3.40.3"},"publisher-location":"Singapore","reference-count":35,"publisher":"Springer Nature Singapore","isbn-type":[{"type":"print","value":"9789819706686"},{"type":"electronic","value":"9789819706693"}],"license":[{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024]]},"DOI":"10.1007\/978-981-97-0669-3_2","type":"book-chapter","created":{"date-parts":[[2024,2,28]],"date-time":"2024-02-28T21:20:16Z","timestamp":1709155216000},"page":"15-26","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Enhancing Visual Question Answering with\u00a0Generated Image Caption"],"prefix":"10.1007","author":[{"given":"Kieu-Anh Thi","family":"Truong","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Truong-Thuy","family":"Tran","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Cam-","family":"Van Thi Nguyen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Duc-Trong","family":"Le","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,2,29]]},"reference":[{"key":"2_CR1","doi-asserted-by":"crossref","unstructured":"Anderson, P., et al.: Bottom-up and top-down attention for image captioning and visual question answering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6077\u20136086 (2018)","DOI":"10.1109\/CVPR.2018.00636"},{"key":"2_CR2","doi-asserted-by":"crossref","unstructured":"Antol, S., et al.: VQA: visual question answering. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2425\u20132433 (2015)","DOI":"10.1109\/ICCV.2015.279"},{"key":"2_CR3","doi-asserted-by":"publisher","first-page":"325","DOI":"10.1016\/j.patrec.2021.09.008","volume":"151","author":"S Barra","year":"2021","unstructured":"Barra, S., Bisogni, C., De Marsico, M., Ricciardi, S.: Visual question answering: which investigated applications? Pattern Recogn. Lett. 151, 325\u2013331 (2021)","journal-title":"Pattern Recogn. Lett."},{"key":"2_CR4","doi-asserted-by":"crossref","unstructured":"Changpinyo, S., Kukliansky, D., Szpektor, I., Chen, X., Ding, N., Soricut, R.: All you may need for VQA are image captions. arXiv preprint: arXiv:2205.01883 (2022)","DOI":"10.18653\/v1\/2022.naacl-main.142"},{"key":"2_CR5","unstructured":"Dinh, H.L., Phan, L.: A jointly language-image model for multilingual visual question answering. In: The 9th International Workshop on Vietnamese Language and Speech Processing (2022)"},{"key":"2_CR6","unstructured":"Dong, N.V.N., Loi: A multi-modal transformer-based method with object prefixes for multilingual visual question answering. In: The 9th International Workshop on Vietnamese Language and Speech Processing (2022)"},{"key":"2_CR7","unstructured":"Dosovitskiy, A., et al.: An image is worth 16x16 words: transformers for image recognition at scale. arXiv preprint: arXiv:2010.11929 (2020)"},{"key":"2_CR8","doi-asserted-by":"crossref","unstructured":"Du, Y., Li, J., Tang, T., Zhao, W.X., Wen, J.R.: Zero-shot visual question answering with language model feedback. arXiv preprint: arXiv:2305.17006 (2023)","DOI":"10.18653\/v1\/2023.findings-acl.590"},{"key":"2_CR9","doi-asserted-by":"crossref","unstructured":"Fukui, A., Park, D.H., Yang, D., Rohrbach, A., Darrell, T., Rohrbach, M.: Multimodal compact bilinear pooling for visual question answering and visual grounding. arXiv preprint: arXiv:1606.01847 (2016)","DOI":"10.18653\/v1\/D16-1044"},{"key":"2_CR10","unstructured":"Gao, H., Mao, J., Zhou, J., Huang, Z., Wang, L., Xu, W.: Are you talking to a machine? Dataset and methods for multilingual image question. In: Advances in Neural Information Processing Systems, vol. 28 (2015)"},{"key":"2_CR11","unstructured":"Huang, Z., Zeng, Z., Liu, B., Fu, D., Fu, J.: Pixel-BERT: aligning image pixels with text by deep multi-modal transformers. arXiv preprint: arXiv:2004.00849 (2020)"},{"key":"2_CR12","doi-asserted-by":"publisher","first-page":"3","DOI":"10.1016\/j.cviu.2017.06.005","volume":"163","author":"K Kafle","year":"2017","unstructured":"Kafle, K., Kanan, C.: Visual question answering: datasets, algorithms, and future challenges. Comput. Vis. Image Underst. 163, 3\u201320 (2017)","journal-title":"Comput. Vis. Image Underst."},{"key":"2_CR13","doi-asserted-by":"crossref","unstructured":"Kim, H., Bansal, M.: Improving visual question answering by referring to generated paragraph captions. arXiv preprint: arXiv:1906.06216 (2019)","DOI":"10.18653\/v1\/P19-1351"},{"key":"2_CR14","unstructured":"Li, C., et al.: SemVLP: vision-language pre-training by aligning semantics at multiple levels. arXiv preprint: arXiv:2103.07829 (2021)"},{"key":"2_CR15","unstructured":"Li, J., Li, D., Xiong, C., Hoi, S.: BLIP: bootstrapping language-image pre-training for unified vision-language understanding and generation. In: International Conference on Machine Learning, pp. 12888\u201312900. PMLR (2022)"},{"key":"2_CR16","doi-asserted-by":"crossref","unstructured":"Li, Q., Fu, J., Yu, D., Mei, T., Luo, J.: Tell-and-answer: towards explainable visual question answering using attributes and captions. arXiv preprint: arXiv:1801.09041 (2018)","DOI":"10.18653\/v1\/D18-1164"},{"key":"2_CR17","series-title":"Lecture Notes in Computer Science()","doi-asserted-by":"publisher","first-page":"552","DOI":"10.1007\/978-3-030-01234-2_34","volume-title":"Computer Vision \u2013 ECCV 2018","author":"Q Li","year":"2018","unstructured":"Li, Q., Tao, Q., Joty, S., Cai, J., Luo, J.: VQA-E: explaining, elaborating, and enhancing your answers for visual questions. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) Computer Vision \u2013 ECCV 2018. Lecture Notes in Computer Science(), vol. 11211, pp. 552\u2013567. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01234-2_34"},{"key":"2_CR18","unstructured":"Lu, J., Batra, D., Parikh, D., Lee, S.: VilBERT: pretraining task-agnostic visiolinguistic representations for vision-and-language tasks. In: Advances in Neural Information Processing Systems, vol. 32 (2019)"},{"key":"2_CR19","doi-asserted-by":"crossref","unstructured":"Malinowski, M., Rohrbach, M., Fritz, M.: Ask your neurons: a neural-based approach to answering questions about images. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 1\u20139 (2015)","DOI":"10.1109\/ICCV.2015.9"},{"key":"2_CR20","doi-asserted-by":"crossref","unstructured":"Pedersoli, M., Lucas, T., Schmid, C., Verbeek, J.: Areas of attention for image captioning. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 1242\u20131250 (2017)","DOI":"10.1109\/ICCV.2017.140"},{"key":"2_CR21","unstructured":"Ren, M., Kiros, R., Zemel, R.: Exploring models and data for image question answering. In: Advances in Neural Information Processing Systems, vol. 28 (2015)"},{"key":"2_CR22","first-page":"1","volume":"81","author":"H Sharma","year":"2021","unstructured":"Sharma, H., Jalal, A.S.: Image captioning improved visual question answering. Multimedia Tools Appl. 81, 1\u201322 (2021)","journal-title":"Multimedia Tools Appl."},{"key":"2_CR23","unstructured":"Singh, J., Ying, V., Nutkiewicz, A.: Attention on attention: architectures for visual question answering (VQA). arXiv preprint: arXiv:1803.07724 (2018)"},{"key":"2_CR24","unstructured":"Su, W., et al.: VL-BERT: pre-training of generic visual-linguistic representations. arXiv preprint: arXiv:1908.08530 (2019)"},{"key":"2_CR25","doi-asserted-by":"crossref","unstructured":"Tan, H., Bansal, M.: LXMERT: learning cross-modality encoder representations from transformers. arXiv preprint: arXiv:1908.07490 (2019)","DOI":"10.18653\/v1\/D19-1514"},{"key":"2_CR26","doi-asserted-by":"crossref","unstructured":"Teney, D., Anderson, P., He, X., Van Den Hengel, A.: Tips and tricks for visual question answering: learnings from the 2017 challenge. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4223\u20134232 (2018)","DOI":"10.1109\/CVPR.2018.00444"},{"key":"2_CR27","doi-asserted-by":"crossref","unstructured":"Thai, T.M., Luu, S.T.: Integrating image features with convolutional sequence-to-sequence network for multilingual visual question answering. arXiv preprint: arXiv:2303.12671 (2023)","DOI":"10.15625\/1813-9663\/18155"},{"key":"2_CR28","doi-asserted-by":"crossref","unstructured":"Tiong, A.M.H., Li, J., Li, B., Savarese, S., Hoi, S.C.: Plug-and-play VQA: Zero-shot VQA by conjoining large pretrained models with zero training. arXiv preprint: arXiv:2210.08773 (2022)","DOI":"10.18653\/v1\/2022.findings-emnlp.67"},{"key":"2_CR29","doi-asserted-by":"crossref","unstructured":"Truong, L.X., Pham, V.Q.: Multi-modal feature extraction for multilingual visual question answering. In: The 9th International Workshop on Vietnamese Language and Speech Processing (2022)","DOI":"10.1142\/S2717554523500108"},{"key":"2_CR30","unstructured":"Wu, J., Chen, L., Mooney, R.J.: Improving VQA and its explanations$$\\backslash $$by comparing competing explanations. arXiv preprint: arXiv:2006.15631 (2020)"},{"key":"2_CR31","unstructured":"Wu, J., Hu, Z., Mooney, R.J.: Joint image captioning and question answering. arXiv preprint: arXiv:1805.08389 (2018)"},{"key":"2_CR32","doi-asserted-by":"crossref","unstructured":"Wu, J., Hu, Z., Mooney, R.J.: Generating question relevant captions to aid visual question answering. arXiv preprint: arXiv:1906.00513 (2019)","DOI":"10.18653\/v1\/P19-1348"},{"issue":"6","key":"2_CR33","doi-asserted-by":"publisher","first-page":"1367","DOI":"10.1109\/TPAMI.2017.2708709","volume":"40","author":"Q Wu","year":"2017","unstructured":"Wu, Q., Shen, C., Wang, P., Dick, A., Van Den Hengel, A.: Image captioning and visual question answering based on attributes and external knowledge. IEEE Trans. Pattern Anal. Mach. Intell. 40(6), 1367\u20131381 (2017)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"2_CR34","doi-asserted-by":"crossref","unstructured":"Yang, Z., et al.: An empirical study of GPT-3 for few-shot knowledge-based VQA. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 36, pp. 3081\u20133089 (2022)","DOI":"10.1609\/aaai.v36i3.20215"},{"key":"2_CR35","doi-asserted-by":"crossref","unstructured":"Yang, Z., He, X., Gao, J., Deng, L., Smola, A.: Stacked attention networks for image question answering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 21\u201329 (2016)","DOI":"10.1109\/CVPR.2016.10"}],"container-title":["Lecture Notes in Computer Science","Computational Data and Social Networks"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-97-0669-3_2","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,13]],"date-time":"2024-11-13T06:29:45Z","timestamp":1731479385000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-97-0669-3_2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024]]},"ISBN":["9789819706686","9789819706693"],"references-count":35,"URL":"https:\/\/doi.org\/10.1007\/978-981-97-0669-3_2","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2024]]},"assertion":[{"value":"29 February 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"CSoNet","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Computational Data and Social Networks","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Hanoi","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Vietnam","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2023","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"11 December 2023","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"13 December 2023","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"12","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"csonet2023","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/csonet-conf.github.io\/csonet23\/index.html","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Single-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Easy Chair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"64","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"23","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"14","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"36% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"2.7","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"2.0","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"The four extended abstracts are also included in this proceedings.","order":10,"name":"additional_info_on_review_process","label":"Additional Info on Review Process","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}