{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,11]],"date-time":"2025-09-11T19:05:42Z","timestamp":1757617542005,"version":"3.44.0"},"reference-count":75,"publisher":"Springer Science and Business Media LLC","issue":"10","license":[{"start":{"date-parts":[[2025,2,4]],"date-time":"2025-02-04T00:00:00Z","timestamp":1738627200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,2,4]],"date-time":"2025-02-04T00:00:00Z","timestamp":1738627200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2020YFC1512601","2020YFC1512601","2020YFC1512601","2020YFC1512601"],"award-info":[{"award-number":["2020YFC1512601","2020YFC1512601","2020YFC1512601","2020YFC1512601"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["62106064","62106064","62106064","62106064"],"award-info":[{"award-number":["62106064","62106064","62106064","62106064"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Vis Comput"],"published-print":{"date-parts":[[2025,8]]},"DOI":"10.1007\/s00371-025-03812-0","type":"journal-article","created":{"date-parts":[[2025,2,4]],"date-time":"2025-02-04T09:39:28Z","timestamp":1738661968000},"page":"7399-7415","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Adaptive sparse triple convolutional attention for enhanced visual question answering"],"prefix":"10.1007","volume":"41","author":[{"given":"Ronggui","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hong","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Juan","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lixia","family":"Xue","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,2,4]]},"reference":[{"key":"3812_CR1","unstructured":"Malinowski, M., Fritz, M.: A multi-world approach to question answering about real-world scenes based on uncertain input. Adv. Neural Inf. Process. Syst. 27 (2014)"},{"key":"3812_CR2","unstructured":"Xu, K., Ba, J., Kiros, R., et\u00a0al.: Show, attend and tell: Neural image caption generation with visual attention. In: International Conference on Machine Learning, PMLR, pp. 2048\u20132057 (2015)"},{"key":"3812_CR3","doi-asserted-by":"crossref","unstructured":"Yu, J., Lu, Y., Qin, Z., et\u00a0al.: Modeling text with graph convolutional network for cross-modal information retrieval. In: Advances in Multimedia Information Processing\u2013PCM 2018: 19th Pacific-Rim Conference on Multimedia, Hefei, China, September 21\u201322, 2018, Proceedings, Part I 19, Springer, pp. 223\u2013234 (2018)","DOI":"10.1007\/978-3-030-00776-8_21"},{"key":"3812_CR4","doi-asserted-by":"crossref","unstructured":"Antol, S., Agrawal, A., Lu, J., et\u00a0al.: Vqa: Visual question answering. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2425\u20132433 (2015)","DOI":"10.1109\/ICCV.2015.279"},{"key":"3812_CR5","doi-asserted-by":"crossref","unstructured":"Yang, Z., He, X., Gao, J., et\u00a0al.: Stacked attention networks for image question answering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 21\u201329 (2016)","DOI":"10.1109\/CVPR.2016.10"},{"key":"3812_CR6","unstructured":"Lu, J., Yang, J., Batra, D., et\u00a0al.: Hierarchical question-image co-attention for visual question answering. Adv. Neural Inf. Process. Syst. 29 (2016)"},{"issue":"1","key":"3812_CR7","doi-asserted-by":"publisher","first-page":"586","DOI":"10.1007\/s10489-022-03559-4","volume":"53","author":"Z Guo","year":"2023","unstructured":"Guo, Z., Han, D.: Sparse co-attention visual question answering networks based on thresholds. Appl. Intell. 53(1), 586\u2013600 (2023)","journal-title":"Appl. Intell."},{"issue":"6","key":"3812_CR8","doi-asserted-by":"publisher","first-page":"1137","DOI":"10.1109\/TPAMI.2016.2577031","volume":"39","author":"S Ren","year":"2016","unstructured":"Ren, S., He, K., Girshick, R., et al.: Faster R-CNN: Towards real-time object detection with region proposal networks. IEEE Trans. Pattern Anal. Mach. Intell. 39(6), 1137\u20131149 (2016)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"3812_CR9","doi-asserted-by":"crossref","unstructured":"Pennington, J., Socher, R., Manning, C.D.: Glove: Global vectors for word representation. In: Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing (EMNLP), pp. 1532\u20131543 (2014)","DOI":"10.3115\/v1\/D14-1162"},{"key":"3812_CR10","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., et\u00a0al.: Attention is all you need. Adv. Neural Inf. Process. Syst. 30 (2017)"},{"key":"3812_CR11","doi-asserted-by":"crossref","unstructured":"Wang, S., Yu, M., Guo, X., et\u00a0al.: R$$^{3}$$: Reinforced ranker-reader for open-domain question answering. In: Proceedings of the AAAI Conference on Artificial Intelligence (2018)","DOI":"10.1609\/aaai.v32i1.12053"},{"key":"3812_CR12","unstructured":"Gao, H., Mao, J., Zhou, J., et\u00a0al.: Are you talking to a machine? Dataset and methods for multilingual image question. Adv. Neural Inf. Process. Syst. 28 (2015)"},{"key":"3812_CR13","doi-asserted-by":"crossref","unstructured":"Patro, B., Namboodiri, V.P.: Differential attention for visual question answering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7680\u20137688 (2018)","DOI":"10.1109\/CVPR.2018.00801"},{"key":"3812_CR14","doi-asserted-by":"publisher","first-page":"6997","DOI":"10.1109\/TMM.2022.3216770","volume":"25","author":"A Mao","year":"2022","unstructured":"Mao, A., Yang, Z., Lin, K., et al.: Positional attention guided transformer-like architecture for visual question answering. IEEE Trans. Multimedia 25, 6997\u20137009 (2022)","journal-title":"IEEE Trans. Multimedia"},{"issue":"13","key":"3812_CR15","doi-asserted-by":"publisher","first-page":"16706","DOI":"10.1007\/s10489-022-04355-w","volume":"53","author":"X Shen","year":"2023","unstructured":"Shen, X., Han, D., Guo, Z., et al.: Local self-attention in transformer for visual question answering. Appl. Intell. 53(13), 16706\u201316723 (2023)","journal-title":"Appl. Intell."},{"key":"3812_CR16","doi-asserted-by":"crossref","unstructured":"Xue, L., Wang, W., Wang, R., et\u00a0al.: Modular dual-stream visual fusion network for visual question answering. Vis. Comput. pp. 1\u201314 (2024)","DOI":"10.1007\/s00371-024-03346-x"},{"key":"3812_CR17","doi-asserted-by":"crossref","unstructured":"Marino, K., Rastegari, M., Farhadi, A., et\u00a0al.: Ok-vqa: A visual question answering benchmark requiring external knowledge. In: Proceedings of the IEEE\/cvf Conference on Computer Vision and Pattern Recognition, pp. 3195\u20133204 (2019)","DOI":"10.1109\/CVPR.2019.00331"},{"key":"3812_CR18","doi-asserted-by":"crossref","unstructured":"Schwenk, D., Khandelwal, A., Clark, C., et\u00a0al.: A-okvqa: A benchmark for visual question answering using world knowledge. In: European Conference on Computer Vision, Springer, pp. 146\u2013162 (2022)","DOI":"10.1007\/978-3-031-20074-8_9"},{"key":"3812_CR19","unstructured":"Yu, Z., Ouyang, X., Shao, Z., et\u00a0al.: Prophet: Prompting large language models with complementary answer heuristics for knowledge-based visual question answering. arXiv preprint arXiv:2303.01903 (2023)"},{"issue":"4","key":"3812_CR20","doi-asserted-by":"publisher","first-page":"280","DOI":"10.1016\/j.vrih.2023.06.002","volume":"6","author":"H Zhang","year":"2024","unstructured":"Zhang, H., Wei, Z., Liu, G., et al.: MKEAH: Multimodal knowledge extraction and accumulation based on hyperplane embedding for knowledge-based visual question answering. Virtual Real. Intell. Hardware 6(4), 280\u2013291 (2024)","journal-title":"Virtual Real. Intell. Hardware"},{"key":"3812_CR21","doi-asserted-by":"crossref","unstructured":"Yuan, L., Chen, Y., Wang, T., et\u00a0al.: Tokens-to-token vit: Training vision transformers from scratch on imagenet. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 558\u2013567 (2021)","DOI":"10.1109\/ICCV48922.2021.00060"},{"key":"3812_CR22","doi-asserted-by":"crossref","unstructured":"Guo, J., Han, K., Wu, H., et\u00a0al.: Cmt: Convolutional neural networks meet vision transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12175\u201312185 (2022)","DOI":"10.1109\/CVPR52688.2022.01186"},{"key":"3812_CR23","unstructured":"Wang, P., Yang, A., Men, R., et\u00a0al.: Ofa: Unifying architectures, tasks, and modalities through a simple sequence-to-sequence learning framework. In: International Conference on Machine Learning, PMLR, pp. 23318\u201323340 (2022)"},{"key":"3812_CR24","doi-asserted-by":"crossref","unstructured":"Li, C., Xu, H., Tian, J., et\u00a0al.: mplug: Effective and efficient vision-language learning by cross-modal skip-connections. arXiv preprint arXiv:2205.12005 (2022)","DOI":"10.18653\/v1\/2022.emnlp-main.488"},{"key":"3812_CR25","unstructured":"Liu, H., Li, C., Wu, Q., et\u00a0al.: Visual instruction tuning. Adv. Neural Inf. Process. Syst. 36 (2024)"},{"key":"3812_CR26","doi-asserted-by":"publisher","first-page":"50","DOI":"10.1109\/TMM.2021.3120873","volume":"25","author":"X Lin","year":"2021","unstructured":"Lin, X., Sun, S., Huang, W., et al.: Eapt: Efficient attention pyramid transformer for image processing. IEEE Trans. Multimedia 25, 50\u201361 (2021)","journal-title":"IEEE Trans. Multimedia"},{"issue":"8","key":"3812_CR27","doi-asserted-by":"publisher","first-page":"4499","DOI":"10.1109\/TNNLS.2021.3116209","volume":"34","author":"Z Xie","year":"2021","unstructured":"Xie, Z., Zhang, W., Sheng, B., et al.: Bagfn: broad attentive graph fusion network for high-order feature interactions. IEEE Trans. Neural Netw. Learn. Syst. 34(8), 4499\u20134513 (2021)","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"3812_CR28","doi-asserted-by":"crossref","unstructured":"Shih, K.J., Singh, S., Hoiem, D.: Where to look: Focus regions for visual question answering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4613\u20134621 (2016)","DOI":"10.1109\/CVPR.2016.499"},{"key":"3812_CR29","doi-asserted-by":"crossref","unstructured":"Yu, Z., Yu, J., Cui, Y., et\u00a0al.: Deep modular co-attention networks for visual question answering. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6281\u20136290 (2019)","DOI":"10.1109\/CVPR.2019.00644"},{"key":"3812_CR30","doi-asserted-by":"crossref","unstructured":"Gao, P., Jiang, Z., You, H., et\u00a0al.: Dynamic fusion with intra-and inter-modality attention flow for visual question answering. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6639\u20136648 (2019)","DOI":"10.1109\/CVPR.2019.00680"},{"key":"3812_CR31","doi-asserted-by":"publisher","first-page":"6730","DOI":"10.1109\/TIP.2021.3097180","volume":"30","author":"W Guo","year":"2021","unstructured":"Guo, W., Zhang, Y., Yang, J., et al.: Re-attention for visual question answering. IEEE Trans. Image Process. 30, 6730\u20136743 (2021)","journal-title":"IEEE Trans. Image Process."},{"key":"3812_CR32","doi-asserted-by":"crossref","unstructured":"Rahman, T., Chou, S.H., Sigal, L., et\u00a0al.: An improved attention for visual question answering. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1653\u20131662 (2021)","DOI":"10.1109\/CVPRW53098.2021.00181"},{"issue":"23","key":"3812_CR33","doi-asserted-by":"publisher","first-page":"6758","DOI":"10.3390\/s20236758","volume":"20","author":"Z Guo","year":"2020","unstructured":"Guo, Z., Han, D.: Multi-modal explicit sparse attention networks for visual question answering. Sensors 20(23), 6758 (2020)","journal-title":"Sensors"},{"issue":"11","key":"3812_CR34","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pone.0277693","volume":"17","author":"X Shen","year":"2022","unstructured":"Shen, X., Han, D., Chen, C., et al.: An effective spatial relational reasoning networks for visual question answering. PLoS ONE 17(11), e0277693 (2022)","journal-title":"PLoS ONE"},{"key":"3812_CR35","doi-asserted-by":"crossref","unstructured":"Hu, J., Shen, L., Sun, G.: Squeeze-and-excitation networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7132\u20137141 (2018)","DOI":"10.1109\/CVPR.2018.00745"},{"key":"3812_CR36","unstructured":"Jaderberg, M., Simonyan, K., Zisserman, A., et\u00a0al.: Spatial transformer networks. Adv. Neural Inf. Process. Syst. 28 (2015)"},{"key":"3812_CR37","unstructured":"Almahairi, A., Ballas, N., Cooijmans, T., et\u00a0al.: Dynamic capacity networks. In: International Conference on Machine Learning, PMLR, pp. 2549\u20132558 (2016)"},{"key":"3812_CR38","doi-asserted-by":"crossref","unstructured":"Anderson, P., He, X., Buehler, C., et\u00a0al.: Bottom-up and top-down attention for image captioning and visual question answering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6077\u20136086 (2018)","DOI":"10.1109\/CVPR.2018.00636"},{"key":"3812_CR39","unstructured":"Zhao, G., Lin, J., Zhang, Z., et\u00a0al.: Explicit sparse transformer: Concentrated attention through explicit selection. arXiv preprint arXiv:1912.11637 (2019)"},{"key":"3812_CR40","doi-asserted-by":"crossref","unstructured":"Teney, D., Anderson, P., He, X., et\u00a0al.: Tips and tricks for visual question answering: Learnings from the 2017 challenge. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4223\u20134232 (2018)","DOI":"10.1109\/CVPR.2018.00444"},{"key":"3812_CR41","doi-asserted-by":"crossref","unstructured":"Goyal, Y., Khot, T., Summers-Stay, D., et\u00a0al.: Making the v in vqa matter: Elevating the role of image understanding in visual question answering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6904\u20136913 (2017)","DOI":"10.1109\/CVPR.2017.670"},{"key":"3812_CR42","doi-asserted-by":"crossref","unstructured":"Hudson, D.A., Manning, C.D.: Gqa: A new dataset for real-world visual reasoning and compositional question answering. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6700\u20136709 (2019)","DOI":"10.1109\/CVPR.2019.00686"},{"key":"3812_CR43","doi-asserted-by":"crossref","unstructured":"Johnson, J., Hariharan, B., Van Der\u00a0Maaten, L., et\u00a0al.: Clevr: A diagnostic dataset for compositional language and elementary visual reasoning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2901\u20132910 (2017)","DOI":"10.1109\/CVPR.2017.215"},{"key":"3812_CR44","unstructured":"Ren, M., Kiros, R., Zemel, R.: Exploring models and data for image question answering. Adv. Neural Inf. Process. Syst. 28 (2015)"},{"key":"3812_CR45","doi-asserted-by":"crossref","unstructured":"Lin, T.Y., Maire, M., Belongie, S., et\u00a0al.: Microsoft coco: Common objects in context. In: Computer Vision\u2013ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6\u201312, 2014, Proceedings, Part V 13, Springer, pp. 740\u2013755 (2014)","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"3812_CR46","unstructured":"Lu, J., Batra, D., Parikh, D., et\u00a0al.: Vilbert: Pretraining task-agnostic visiolinguistic representations for vision-and-language tasks. Adv. Neural Inf. Process. Syst. 32 (2019)"},{"key":"3812_CR47","unstructured":"Li, L.H., Yatskar, M., Yin, D., et\u00a0al.: Visualbert: A simple and performant baseline for vision and language. arXiv preprint arXiv:1908.03557 (2019)"},{"key":"3812_CR48","unstructured":"Su, W., Zhu, X., Cao, Y., et\u00a0al.: Vl-bert: Pre-training of generic visual-linguistic representations. arXiv preprint arXiv:1908.08530 (2019)"},{"key":"3812_CR49","doi-asserted-by":"crossref","unstructured":"Tan, H., Bansal, M.: Lxmert: Learning cross-modality encoder representations from transformers. arXiv preprint arXiv:1908.07490(2019)","DOI":"10.18653\/v1\/D19-1514"},{"key":"3812_CR50","doi-asserted-by":"crossref","unstructured":"Chen, Y.C., Li, L., Yu, L., et\u00a0al.: Uniter: Universal image-text representation learning. In: European Conference on Computer Vision, Springer, pp. 104\u2013120 (2020)","DOI":"10.1007\/978-3-030-58577-8_7"},{"key":"3812_CR51","unstructured":"Li, J., Li, D., Savarese, S., et\u00a0al.: Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models. In: International Conference on Machine Learning, PMLR, pp. 19730\u201319742 (2023)"},{"key":"3812_CR52","unstructured":"Yu, J., Wang, Z., Vasudevan, V., et\u00a0al.: Coca: Contrastive captioners are image-text foundation models. arXiv preprint arXiv:2205.01917 (2022)"},{"key":"3812_CR53","doi-asserted-by":"crossref","unstructured":"Wang, W., Bao, H., Dong, L., et\u00a0al.: Image as a foreign language: Beit pretraining for all vision and vision-language tasks. arXiv preprint arXiv:2208.10442 (2022)","DOI":"10.1109\/CVPR52729.2023.01838"},{"key":"3812_CR54","doi-asserted-by":"crossref","unstructured":"Cadene, R., Ben-Younes, H., Cord, M., et\u00a0al.: Murel: Multimodal relational reasoning for visual question answering. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1989\u20131998 (2019)","DOI":"10.1109\/CVPR.2019.00209"},{"key":"3812_CR55","doi-asserted-by":"crossref","unstructured":"Li, L., Gan, Z., Cheng, Y., et\u00a0al.: Relation-aware graph attention network for visual question answering. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 10313\u201310322 (2019)","DOI":"10.1109\/ICCV.2019.01041"},{"key":"3812_CR56","unstructured":"Yu, Z., Cui, Y., Yu, J., et\u00a0al.: Multimodal unified attention networks for vision-and-language interactions. arXiv preprint arXiv:1908.04107 (2019)"},{"key":"3812_CR57","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2023.110706","volume":"275","author":"C Chen","year":"2023","unstructured":"Chen, C., Han, D., Shen, X.: CLVIN: Complete language-vision interaction network for visual question answering. Knowl.-Based Syst. 275, 110706 (2023)","journal-title":"Knowl.-Based Syst."},{"key":"3812_CR58","doi-asserted-by":"publisher","first-page":"35662","DOI":"10.1109\/ACCESS.2020.2975093","volume":"8","author":"C Chen","year":"2020","unstructured":"Chen, C., Han, D., Wang, J.: Multimodal encoder-decoder attention networks for visual question answering. Ieee Access 8, 35662\u201335671 (2020)","journal-title":"Ieee Access"},{"key":"3812_CR59","unstructured":"Hudson, D.A., Manning, C.D.: Compositional attention networks for machine reasoning. arXiv preprint arXiv:1803.03067 (2018)"},{"key":"3812_CR60","unstructured":"Kim, J.H., Jun, J., Zhang, B.T.: Bilinear attention networks. Adv. Neural Inf. Process. Syst. 31 (2018)"},{"key":"3812_CR61","doi-asserted-by":"crossref","unstructured":"Miao, Y., Cheng, W., He, S., et\u00a0al.: Research on visual question answering based on gat relational reasoning. Neural Process. Lett. pp. 1\u201314 (2022)","DOI":"10.1007\/s11063-021-10689-2"},{"issue":"5","key":"3812_CR62","doi-asserted-by":"publisher","first-page":"2527","DOI":"10.1007\/s00530-023-01125-7","volume":"29","author":"C Liu","year":"2023","unstructured":"Liu, C., Tan, Y.Y., Xia, T.T., et al.: Co-attention graph convolutional network for visual question answering. Multimedia Syst. 29(5), 2527\u20132543 (2023)","journal-title":"Multimedia Syst."},{"key":"3812_CR63","doi-asserted-by":"publisher","first-page":"4282","DOI":"10.1109\/TMM.2022.3173131","volume":"25","author":"B Qin","year":"2022","unstructured":"Qin, B., Hu, H., Zhuang, Y.: Deep residual weight-sharing attention network with low-rank attention for visual question answering. IEEE Trans. Multimedia 25, 4282\u20134295 (2022)","journal-title":"IEEE Trans. Multimedia"},{"key":"3812_CR64","doi-asserted-by":"crossref","unstructured":"Zhou, Y., Ren, T., Zhu, C., et\u00a0al.: Trar: Routing the attention spans in transformer for visual question answering. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2074\u20132084 (2021)","DOI":"10.1109\/ICCV48922.2021.00208"},{"key":"3812_CR65","unstructured":"Gulrajani, I., Ahmed, F., Arjovsky, M., et\u00a0al.: Improved training of wasserstein gans. Adv. Neural Inf. Process. Syst. 30 (2017)"},{"key":"3812_CR66","doi-asserted-by":"crossref","unstructured":"Mascharka, D., Tran, P., Soklaski, R., et\u00a0al.: Transparency by design: Closing the gap between performance and interpretability in visual reasoning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4942\u20134950 (2018)","DOI":"10.1109\/CVPR.2018.00519"},{"key":"3812_CR67","doi-asserted-by":"publisher","first-page":"1264","DOI":"10.1109\/TMM.2020.2995278","volume":"23","author":"H Zhong","year":"2020","unstructured":"Zhong, H., Chen, J., Shen, C., et al.: Self-adaptive neural module transformer for visual question answering. IEEE Trans. Multimedia 23, 1264\u20131273 (2020)","journal-title":"IEEE Trans. Multimedia"},{"issue":"12","key":"3812_CR68","doi-asserted-by":"publisher","first-page":"3196","DOI":"10.1109\/TMM.2020.2972830","volume":"22","author":"J Yu","year":"2020","unstructured":"Yu, J., Zhang, W., Lu, Y., et al.: Reasoning on the relation: Enhancing visual representation for visual question answering and cross-modal retrieval. IEEE Trans. Multimedia 22(12), 3196\u20133209 (2020)","journal-title":"IEEE Trans. Multimedia"},{"issue":"1","key":"3812_CR69","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/s11263-023-01871-1","volume":"132","author":"G Luo","year":"2024","unstructured":"Luo, G., Zhou, Y., Sun, X., et al.: Towards language-guided visual recognition via dynamic convolutions. Int. J. Comput. Vision 132(1), 1\u201319 (2024)","journal-title":"Int. J. Comput. Vision"},{"key":"3812_CR70","doi-asserted-by":"publisher","DOI":"10.1016\/j.jvcir.2020.102762","volume":"73","author":"B Sun","year":"2020","unstructured":"Sun, B., Yao, Z., Zhang, Y., et al.: Local relation network with multilevel attention for visual question answering. J. Vis. Commun. Image Represent. 73, 102762 (2020)","journal-title":"J. Vis. Commun. Image Represent."},{"key":"3812_CR71","doi-asserted-by":"crossref","unstructured":"Wu, C., Liu, J., Wang, X., et\u00a0al.: Object-difference attention: A simple relational attention for visual question answering. In: Proceedings of the 26th ACM International Conference on Multimedia, pp. 519\u2013527 (2018)","DOI":"10.1145\/3240508.3240513"},{"key":"3812_CR72","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2021.107956","volume":"117","author":"Y Liu","year":"2021","unstructured":"Liu, Y., Zhang, X., Zhang, Q., et al.: Dual self-attention with co-attention networks for visual question answering. Pattern Recogn. 117, 107956 (2021)","journal-title":"Pattern Recogn."},{"issue":"6","key":"3812_CR73","doi-asserted-by":"publisher","first-page":"4520","DOI":"10.1109\/TCYB.2020.3029423","volume":"52","author":"Y Liu","year":"2020","unstructured":"Liu, Y., Zhang, X., Zhao, Z., et al.: ALSA: adversarial learning of supervised attentions for visual question answering. IEEE Trans. Cybernet. 52(6), 4520\u20134533 (2020)","journal-title":"IEEE Trans. Cybernet."},{"key":"3812_CR74","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2022.108878","volume":"131","author":"K Shuang","year":"2022","unstructured":"Shuang, K., Guo, J., Wang, Z.: Comprehensive-perception dynamic reasoning for visual question answering. Pattern Recogn. 131, 108878 (2022)","journal-title":"Pattern Recogn."},{"issue":"18","key":"3812_CR75","doi-asserted-by":"publisher","first-page":"20967","DOI":"10.1007\/s10489-023-04564-x","volume":"53","author":"H Xia","year":"2023","unstructured":"Xia, H., Lan, R., Li, H., et al.: ST-VQA: shrinkage transformer with accurate alignment for visual question answering. Appl. Intell. 53(18), 20967\u201320978 (2023)","journal-title":"Appl. Intell."}],"container-title":["The Visual Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-025-03812-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00371-025-03812-0\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-025-03812-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,6]],"date-time":"2025-09-06T04:45:51Z","timestamp":1757133951000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00371-025-03812-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,2,4]]},"references-count":75,"journal-issue":{"issue":"10","published-print":{"date-parts":[[2025,8]]}},"alternative-id":["3812"],"URL":"https:\/\/doi.org\/10.1007\/s00371-025-03812-0","relation":{},"ISSN":["0178-2789","1432-2315"],"issn-type":[{"type":"print","value":"0178-2789"},{"type":"electronic","value":"1432-2315"}],"subject":[],"published":{"date-parts":[[2025,2,4]]},"assertion":[{"value":"10 January 2025","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 February 2025","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no conflict of interest to declare that are relevant to the content of this article.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"All procedures performed in studies involving human participants were in accordance with the ethical standards of the institutional and\/or national research committee and with the 1964 Helsinki declaration and its later amendments or comparable ethical standards. Informed consent was obtained from all individual participants included in the study.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical Approval"}}]}}