{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,14]],"date-time":"2026-05-14T23:03:28Z","timestamp":1778799808199,"version":"3.51.4"},"reference-count":45,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2026,1,7]],"date-time":"2026-01-07T00:00:00Z","timestamp":1767744000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,7]],"date-time":"2026-01-07T00:00:00Z","timestamp":1767744000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"National Key Research and Development Program of China,China","award":["2020YFC1512601"],"award-info":[{"award-number":["2020YFC1512601"]}]},{"name":"National Key Research and Development Program of China,China","award":["2020YFC1512601"],"award-info":[{"award-number":["2020YFC1512601"]}]},{"name":"National Key Research and Development Program of China,China","award":["2020YFC1512601"],"award-info":[{"award-number":["2020YFC1512601"]}]},{"name":"National Key Research and Development Program of China,China","award":["2020YFC1512601"],"award-info":[{"award-number":["2020YFC1512601"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China,China","doi-asserted-by":"crossref","award":["62106064"],"award-info":[{"award-number":["62106064"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China,China","doi-asserted-by":"crossref","award":["62106064"],"award-info":[{"award-number":["62106064"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China,China","doi-asserted-by":"crossref","award":["62106064"],"award-info":[{"award-number":["62106064"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China,China","doi-asserted-by":"crossref","award":["62106064"],"award-info":[{"award-number":["62106064"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Multimed Info Retr"],"published-print":{"date-parts":[[2026,3]]},"DOI":"10.1007\/s13735-025-00394-4","type":"journal-article","created":{"date-parts":[[2026,1,7]],"date-time":"2026-01-07T09:41:16Z","timestamp":1767778876000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["CSAM: Capsule spatial attention mask network for visual question answering"],"prefix":"10.1007","volume":"15","author":[{"given":"Lixia","family":"Xue","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yi","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ronggui","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Juan","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,1,7]]},"reference":[{"key":"394_CR1","unstructured":"Vaswani A (2017) Attention is all you need. Advances in Neural Information Processing Systems"},{"key":"394_CR2","doi-asserted-by":"crossref","unstructured":"Yu Z, Yu J, Cui Y, et\u00a0al (2019) Deep modular co-attention networks for visual question answering. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 6281\u20136290","DOI":"10.1109\/CVPR.2019.00644"},{"key":"394_CR3","doi-asserted-by":"crossref","unstructured":"Gao P, Jiang Z, You H, et\u00a0al (2019) Dynamic fusion with intra-and inter-modality attention flow for visual question answering. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 6639\u20136648","DOI":"10.1109\/CVPR.2019.00680"},{"key":"394_CR4","doi-asserted-by":"crossref","unstructured":"Luo Y, Ji J, Sun X, et\u00a0al (2021) Dual-level collaborative transformer for image captioning. In: Proceedings of the AAAI conference on artificial intelligence, pp 2286\u20132293","DOI":"10.1609\/aaai.v35i3.16328"},{"key":"394_CR5","doi-asserted-by":"crossref","unstructured":"Zhang X, Sun X, Luo Y, et\u00a0al (2021) Rstnet: Captioning with adaptive attention on visual and non-visual words. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 15465\u201315474","DOI":"10.1109\/CVPR46437.2021.01521"},{"key":"394_CR6","unstructured":"Radford A, Kim JW, Hallacy C, et\u00a0al (2021) Learning transferable visual models from natural language supervision. In: International conference on machine learning, PMLR, pp 8748\u20138763"},{"issue":"12","key":"394_CR7","doi-asserted-by":"publisher","first-page":"5947","DOI":"10.1109\/TNNLS.2018.2817340","volume":"29","author":"Z Yu","year":"2018","unstructured":"Yu Z, Yu J, Xiang C et al (2018) Beyond bilinear: Generalized multimodal factorized high-order pooling for visual question answering. IEEE transactions on neural networks and learning systems 29(12):5947\u20135959","journal-title":"IEEE transactions on neural networks and learning systems"},{"key":"394_CR8","unstructured":"Kim JH, Jun J, Zhang BT (2018) Bilinear attention networks. Advances in neural information processing systems 31"},{"key":"394_CR9","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2023.109420","volume":"138","author":"Y Ma","year":"2023","unstructured":"Ma Y, Ji J, Sun X et al (2023) Towards local visual modeling for image captioning. Pattern Recogn 138:109420","journal-title":"Pattern Recogn"},{"issue":"13","key":"394_CR10","doi-asserted-by":"publisher","first-page":"16706","DOI":"10.1007\/s10489-022-04355-w","volume":"53","author":"X Shen","year":"2023","unstructured":"Shen X, Han D, Guo Z et al (2023) Local self-attention in transformer for visual question answering. Appl Intell 53(13):16706\u201316723","journal-title":"Appl Intell"},{"key":"394_CR11","doi-asserted-by":"crossref","unstructured":"Jiang H, Misra I, Rohrbach M, et\u00a0al (2020) In defense of grid features for visual question answering. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 10267\u201310276","DOI":"10.1109\/CVPR42600.2020.01028"},{"issue":"23","key":"394_CR12","doi-asserted-by":"publisher","first-page":"6758","DOI":"10.3390\/s20236758","volume":"20","author":"Z Guo","year":"2020","unstructured":"Guo Z, Han D (2020) Multi-modal explicit sparse attention networks for visual question answering. Sensors 20(23):6758","journal-title":"Sensors"},{"issue":"1","key":"394_CR13","doi-asserted-by":"publisher","first-page":"586","DOI":"10.1007\/s10489-022-03559-4","volume":"53","author":"Z Guo","year":"2023","unstructured":"Guo Z, Han D (2023) Sparse co-attention visual question answering networks based on thresholds. Appl Intell 53(1):586\u2013600","journal-title":"Appl Intell"},{"key":"394_CR14","unstructured":"Radford A, Kim JW, Hallacy C, et\u00a0al (2021) Learning transferable visual models from natural language supervision. In: International conference on machine learning, PMLR, pp 8748\u20138763"},{"key":"394_CR15","doi-asserted-by":"crossref","unstructured":"Nam H, Ha JW, Kim J (2017) Dual attention networks for multimodal reasoning and matching. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 299\u2013307","DOI":"10.1109\/CVPR.2017.232"},{"key":"394_CR16","doi-asserted-by":"crossref","unstructured":"Fan H, Zhou J (2018) Stacked latent attention for multimodal reasoning. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 1072\u20131080","DOI":"10.1109\/CVPR.2018.00118"},{"issue":"11","key":"394_CR17","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pone.0277693","volume":"17","author":"X Shen","year":"2022","unstructured":"Shen X, Han D, Chen C et al (2022) An effective spatial relational reasoning networks for visual question answering. PLoS ONE 17(11):e0277693","journal-title":"PLoS ONE"},{"key":"394_CR18","doi-asserted-by":"publisher","first-page":"6997","DOI":"10.1109\/TMM.2022.3216770","volume":"25","author":"A Mao","year":"2022","unstructured":"Mao A, Yang Z, Lin K et al (2022) Positional attention guided transformer-like architecture for visual question answering. IEEE Trans Multimedia 25:6997\u20137009","journal-title":"IEEE Trans Multimedia"},{"key":"394_CR19","doi-asserted-by":"publisher","first-page":"6730","DOI":"10.1109\/TIP.2021.3097180","volume":"30","author":"W Guo","year":"2021","unstructured":"Guo W, Zhang Y, Yang J et al (2021) Re-attention for visual question answering. IEEE Trans Image Process 30:6730\u20136743","journal-title":"IEEE Trans Image Process"},{"key":"394_CR20","doi-asserted-by":"crossref","unstructured":"Yang Z, He X, Gao J, et\u00a0al (2016) Stacked attention networks for image question answering. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 21\u201329","DOI":"10.1109\/CVPR.2016.10"},{"key":"394_CR21","doi-asserted-by":"crossref","unstructured":"Cui Y, Ren W, Cao X, et\u00a0al (2023) Focal network for image restoration. 2023 IEEE\/CVF International Conference on Computer Vision (ICCV) pp 12955\u201312965","DOI":"10.1109\/ICCV51070.2023.01195"},{"key":"394_CR22","doi-asserted-by":"crossref","unstructured":"Cui Y, Ren W, Knoll A (2024) Omni-kernel network for image restoration. In: AAAI Conference on Artificial Intelligence","DOI":"10.1609\/aaai.v38i2.27907"},{"key":"394_CR23","doi-asserted-by":"crossref","unstructured":"Cui Y, Tao Y, Bing Z, et\u00a0al (2023) Selective frequency network for image restoration. In: International Conference on Learning Representations","DOI":"10.1109\/ICCV51070.2023.01195"},{"key":"394_CR24","doi-asserted-by":"publisher","first-page":"5735","DOI":"10.1109\/LRA.2023.3300254","volume":"8","author":"Y Cui","year":"2023","unstructured":"Cui Y, Knoll A (2023) Psnet: Towards efficient image restoration with self-attention. IEEE Robotics and Automation Letters 8:5735\u20135742","journal-title":"IEEE Robotics and Automation Letters"},{"key":"394_CR25","doi-asserted-by":"crossref","unstructured":"Guo J, Han K, Wu H, et\u00a0al (2022) Cmt: Convolutional neural networks meet vision transformers. 2022 ieee. In: CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp 12165\u201312175","DOI":"10.1109\/CVPR52688.2022.01186"},{"key":"394_CR26","first-page":"15908","volume":"34","author":"K Han","year":"2021","unstructured":"Han K, Xiao A, Wu E et al (2021) Transformer in transformer. Adv Neural Inf Process Syst 34:15908\u201315919","journal-title":"Adv Neural Inf Process Syst"},{"key":"394_CR27","doi-asserted-by":"crossref","unstructured":"Zhou Y, Ren T, Zhu C, et\u00a0al (2021) Trar: Routing the attention spans in transformer for visual question answering. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 2074\u20132084","DOI":"10.1109\/ICCV48922.2021.00208"},{"key":"394_CR28","doi-asserted-by":"publisher","first-page":"32","DOI":"10.1007\/s11263-016-0981-7","volume":"123","author":"R Krishna","year":"2017","unstructured":"Krishna R, Zhu Y, Groth O et al (2017) Visual genome: Connecting language and vision using crowdsourced dense image annotations. Int J Comput Vision 123:32\u201373","journal-title":"Int J Comput Vision"},{"key":"394_CR29","unstructured":"Sabour S, Frosst N, Hinton GE (2017) Dynamic routing between capsules. Advances in neural information processing systems 30"},{"key":"394_CR30","doi-asserted-by":"crossref","unstructured":"Teney D, Anderson P, He X, et\u00a0al (2018) Tips and tricks for visual question answering: Learnings from the 2017 challenge. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 4223\u20134232","DOI":"10.1109\/CVPR.2018.00444"},{"key":"394_CR31","doi-asserted-by":"crossref","unstructured":"Antol S, Agrawal A, Lu J, et\u00a0al (2015) Vqa: Visual question answering. In: Proceedings of the IEEE international conference on computer vision, pp 2425\u20132433","DOI":"10.1109\/ICCV.2015.279"},{"key":"394_CR32","doi-asserted-by":"crossref","unstructured":"Johnson J, Hariharan B, Van Der\u00a0Maaten L, et\u00a0al (2017) Clevr: A diagnostic dataset for compositional language and elementary visual reasoning. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 2901\u20132910","DOI":"10.1109\/CVPR.2017.215"},{"key":"394_CR33","unstructured":"Kinga D, Adam JB, et\u00a0al (2015) A method for stochastic optimization. In: International conference on learning representations (ICLR), San Diego, California;, p\u00a06"},{"key":"394_CR34","doi-asserted-by":"crossref","unstructured":"Nguyen DK, Okatani T (2018) Improved fusion of visual and language representations by dense symmetric co-attention for visual question answering. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 6087\u20136096","DOI":"10.1109\/CVPR.2018.00637"},{"key":"394_CR35","doi-asserted-by":"crossref","unstructured":"Liu H, Han D, Zhang S, et\u00a0al (2025) Asam: Asynchronous self-attention model for visual question answering. Computer Science and Information Systems","DOI":"10.2298\/CSIS240321003L"},{"key":"394_CR36","doi-asserted-by":"publisher","first-page":"350","DOI":"10.1007\/s00530-024-01568-6","volume":"30","author":"J Shi","year":"2024","unstructured":"Shi J, Han D, Chen C et al (2024) Ktmn: Knowledge-driven two-stage modulation network for visual question answering. Multim Syst 30:350","journal-title":"Multim Syst"},{"key":"394_CR37","doi-asserted-by":"crossref","unstructured":"Song X, Han D, Chen C, et\u00a0al (2024) Vman: visual-modified attention network for multimodal paradigms. The Visual Computer","DOI":"10.1007\/s00371-024-03563-4"},{"key":"394_CR38","doi-asserted-by":"crossref","unstructured":"Yi J, Han D, Chen C, et\u00a0al (2024) Ardn: Attention re-distribution network for visual question answering. Arabian Journal for Science and Engineering","DOI":"10.1007\/s13369-024-09067-6"},{"key":"394_CR39","doi-asserted-by":"crossref","unstructured":"Shi J, Han D, Chen C, et\u00a0al (2025) Saffnet: self-attention based on fourier frequency domain filter network for visual question answering. The Visual Computer","DOI":"10.1007\/s00371-024-03777-6"},{"key":"394_CR40","unstructured":"Gulrajani I, Ahmed F, Arjovsky M, et\u00a0al (2017) Improved training of wasserstein gans. Advances in neural information processing systems 30"},{"key":"394_CR41","doi-asserted-by":"crossref","unstructured":"Mascharka D, Tran P, Soklaski R, et\u00a0al (2018) Transparency by design: Closing the gap between performance and interpretability in visual reasoning. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 4942\u20134950","DOI":"10.1109\/CVPR.2018.00519"},{"key":"394_CR42","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2023.110084","volume":"147","author":"C Chen","year":"2023","unstructured":"Chen C, Han D, Chang CC (2023) Mpcct: Multimodal vision-language learning paradigm with context-based compact transformer. Pattern Recognit 147:110084","journal-title":"Pattern Recognit"},{"key":"394_CR43","doi-asserted-by":"crossref","unstructured":"Bao Y, Xing T, Chen X (2023) Confidence-based interactable neural-symbolic visual question answering. Neurocomputing 564:126991","DOI":"10.1016\/j.neucom.2023.126991"},{"key":"394_CR44","doi-asserted-by":"crossref","unstructured":"Wang M, Yang J, Xue L, et\u00a0al (2024) Joat: to dynamically aggregate visual queries in transformer for visual question answering. Third International Conference on Machine Vision, Automatic Identification, and Detection (MVAID 2024)","DOI":"10.1117\/12.3035665"},{"key":"394_CR45","doi-asserted-by":"crossref","unstructured":"Shi J, Zhang H, Li J (2019) Explainable and explicit visual reasoning over scene graphs. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 8376\u20138384","DOI":"10.1109\/CVPR.2019.00857"}],"container-title":["International Journal of Multimedia Information Retrieval"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13735-025-00394-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s13735-025-00394-4","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13735-025-00394-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,16]],"date-time":"2026-03-16T11:10:27Z","timestamp":1773659427000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s13735-025-00394-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,1,7]]},"references-count":45,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2026,3]]}},"alternative-id":["394"],"URL":"https:\/\/doi.org\/10.1007\/s13735-025-00394-4","relation":{},"ISSN":["2192-6611","2192-662X"],"issn-type":[{"value":"2192-6611","type":"print"},{"value":"2192-662X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,1,7]]},"assertion":[{"value":"8 December 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 February 2025","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 December 2025","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"7 January 2026","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"All procedures performed in studies involving human participants were in accordance with the ethical standards of the institutional and\/or national research committee and with the 1964 Helsinki declaration and its later amendments or comparable ethical standards. Informed consent was obtained from all individual participants included in the study.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical Approval"}},{"value":"The authors declare no competing interests.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing Interests"}}],"article-number":"2"}}