{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,25]],"date-time":"2025-03-25T18:15:46Z","timestamp":1742926546064,"version":"3.40.3"},"publisher-location":"Cham","reference-count":17,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783031162091"},{"type":"electronic","value":"9783031162107"}],"license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022]]},"DOI":"10.1007\/978-3-031-16210-7_35","type":"book-chapter","created":{"date-parts":[[2022,9,20]],"date-time":"2022-09-20T23:03:09Z","timestamp":1663714989000},"page":"423-435","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["RVT-Transformer: Residual Attention in\u00a0Answerability Prediction on\u00a0Visual Question Answering for\u00a0Blind People"],"prefix":"10.1007","author":[{"given":"Duy-Minh","family":"Nguyen-Tran","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tung","family":"Le","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Khoa","family":"Pho","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Minh Le","family":"Nguyen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Huy Tien","family":"Nguyen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,9,21]]},"reference":[{"key":"35_CR1","doi-asserted-by":"publisher","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: BERT: pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers), pp. 4171\u20134186. Association for Computational Linguistics, Minneapolis, Minnesota (2019). https:\/\/doi.org\/10.18653\/v1\/N19-1423, https:\/\/aclanthology.org\/N19-1423","DOI":"10.18653\/v1\/N19-1423"},{"key":"35_CR2","unstructured":"Dosovitskiy, A., et al.: An image is worth 16x16 words: transformers for image recognition at scale. In: International Conference on Learning Representations (2021)"},{"issue":"4","key":"35_CR3","doi-asserted-by":"publisher","first-page":"398","DOI":"10.1007\/s11263-018-1116-0","volume":"127","author":"Y Goyal","year":"2019","unstructured":"Goyal, Y., Khot, T., Agrawal, A., Summers-Stay, D., Batra, D., Parikh, D.: Making the V in VQA matter: elevating the role of image understanding in visual question answering. Int. J. Comput. Vis. 127(4), 398\u2013414 (2019)","journal-title":"Int. J. Comput. Vis."},{"key":"35_CR4","doi-asserted-by":"publisher","unstructured":"Gurari, D., et al.: Vizwiz-priv: a dataset for recognizing the presence and purpose of private visual information in images taken by blind people. In: 2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 939\u2013948 (2019). https:\/\/doi.org\/10.1109\/CVPR.2019.00103","DOI":"10.1109\/CVPR.2019.00103"},{"key":"35_CR5","doi-asserted-by":"crossref","unstructured":"Gurari, D., et al.: Vizwiz grand challenge: answering visual questions from blind people. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2018)","DOI":"10.1109\/CVPR.2018.00380"},{"key":"35_CR6","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"35_CR7","doi-asserted-by":"publisher","first-page":"366","DOI":"10.1016\/j.neucom.2020.03.098","volume":"402","author":"J Hong","year":"2020","unstructured":"Hong, J., Park, S., Byun, H.: Selective residual learning for visual question answering. Neurocomputing 402, 366\u2013374 (2020). https:\/\/doi.org\/10.1016\/j.neucom.2020.03.098. www.sciencedirect.com\/science\/article\/pii\/S0925231220304859","journal-title":"Neurocomputing"},{"issue":"4","key":"35_CR8","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3460474","volume":"17","author":"W Jiang","year":"2021","unstructured":"Jiang, W., Wang, W., Hu, H.: Bi-directional co-attention network for image captioning. ACM Trans. Multimedia Comput. Commun. Appl. 17(4), 1\u201320 (2021). https:\/\/doi.org\/10.1145\/3460474","journal-title":"ACM Trans. Multimedia Comput. Commun. Appl."},{"key":"35_CR9","unstructured":"Kazemi, V., Elqursh, A.: Show, ask, attend, and answer: a strong baseline for visual question answering. arXiv preprint arXiv:1704.03162 (2017)"},{"key":"35_CR10","doi-asserted-by":"publisher","first-page":"451","DOI":"10.1016\/j.neucom.2021.08.117","volume":"465","author":"T Le","year":"2021","unstructured":"Le, T., Nguyen, H.T., Nguyen, M.L.: Multi visual and textual embedding on visual question answering for blind people. Neurocomputing 465, 451\u2013464 (2021). https:\/\/doi.org\/10.1016\/j.neucom.2021.08.117","journal-title":"Neurocomputing"},{"key":"35_CR11","doi-asserted-by":"publisher","unstructured":"Le, T., Nguyen, H.T., Nguyen, M.L.: Vision and text transformer for predicting answerability on visual question answering. In: 2021 IEEE International Conference on Image Processing (ICIP), pp. 934\u2013938 (2021). https:\/\/doi.org\/10.1109\/ICIP42928.2021.9506796","DOI":"10.1109\/ICIP42928.2021.9506796"},{"key":"35_CR12","doi-asserted-by":"publisher","unstructured":"Le., T., Pho., K., Bui., T., Nguyen., H.T., Nguyen., M.L.: Object-less vision-language model on visual question classification for blind people. In: Proceedings of the 14th International Conference on Agents and Artificial Intelligence - Volume 3: ICAART, pp. 180\u2013187. INSTICC, SciTePress (2022). https:\/\/doi.org\/10.5220\/0010797400003116","DOI":"10.5220\/0010797400003116"},{"key":"35_CR13","doi-asserted-by":"publisher","unstructured":"Le, T., Tien Huy, N., Le Minh, N.: Integrating transformer into global and residual image feature extractor in visual question answering for blind people. In: 2020 12th International Conference on Knowledge and Systems Engineering (KSE), pp. 31\u201336 (2020). https:\/\/doi.org\/10.1109\/KSE50997.2020.9287539","DOI":"10.1109\/KSE50997.2020.9287539"},{"key":"35_CR14","unstructured":"Simonyan, K., Zisserman, A.: Very deep convolutional networks for large-scale image recognition. In: International Conference on Learning Representations (2015)"},{"key":"35_CR15","doi-asserted-by":"crossref","unstructured":"Tan, H., Bansal, M.: Lxmert: learning cross-modality encoder representations from transformers (2019)","DOI":"10.18653\/v1\/D19-1514"},{"key":"35_CR16","doi-asserted-by":"crossref","unstructured":"Wang, T., Huang, J., Zhang, H., Sun, Q.: Visual commonsense R-CNN. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2020)","DOI":"10.1109\/CVPR42600.2020.01077"},{"key":"35_CR17","doi-asserted-by":"publisher","first-page":"2986","DOI":"10.1109\/TMM.2021.3091882","volume":"24","author":"X Zhang","year":"2021","unstructured":"Zhang, X., Zhang, F., Xu, C.: Explicit cross-modal representation learning for visual commonsense reasoning. IEEE Trans. Multimedia 24, 2986\u20132997 (2021). https:\/\/doi.org\/10.1109\/TMM.2021.3091882","journal-title":"IEEE Trans. Multimedia"}],"container-title":["Communications in Computer and Information Science","Advances in Computational Collective Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-16210-7_35","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,3,9]],"date-time":"2023-03-09T12:21:34Z","timestamp":1678364494000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-16210-7_35"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022]]},"ISBN":["9783031162091","9783031162107"],"references-count":17,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-16210-7_35","relation":{},"ISSN":["1865-0929","1865-0937"],"issn-type":[{"type":"print","value":"1865-0929"},{"type":"electronic","value":"1865-0937"}],"subject":[],"published":{"date-parts":[[2022]]},"assertion":[{"value":"21 September 2022","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}}]}}