{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,8]],"date-time":"2026-08-08T00:42:53Z","timestamp":1786149773149,"version":"3.56.0"},"reference-count":63,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U23A20389"],"award-info":[{"award-number":["U23A20389"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62306095"],"award-info":[{"award-number":["62306095"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["82441008"],"award-info":[{"award-number":["82441008"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62506101"],"award-info":[{"award-number":["62506101"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["82441009"],"award-info":[{"award-number":["82441009"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100004027","name":"Postdoctoral Foundation of Hei Long Jiang Province","doi-asserted-by":"publisher","award":["LBH-Z24180"],"award-info":[{"award-number":["LBH-Z24180"]}],"id":[{"id":"10.13039\/501100004027","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100005046","name":"Natural Science Foundation of Heilongjiang Province","doi-asserted-by":"publisher","award":["ZD2022F002"],"award-info":[{"award-number":["ZD2022F002"]}],"id":[{"id":"10.13039\/501100005046","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100002858","name":"China Postdoctoral Science Foundation","doi-asserted-by":"publisher","award":["2024M764190"],"award-info":[{"award-number":["2024M764190"]}],"id":[{"id":"10.13039\/501100002858","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Neurocomputing"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1016\/j.neucom.2026.134131","type":"journal-article","created":{"date-parts":[[2026,6,3]],"date-time":"2026-06-03T16:18:23Z","timestamp":1780503503000},"page":"134131","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Looking closer and smarter: Multi-scale progressive attention for visual text question answering"],"prefix":"10.1016","volume":"697","author":[{"given":"Kang","family":"Chen","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiangqian","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.neucom.2026.134131_bib0005","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"19175","article-title":"Image as a foreign language: BEiT pretraining for vision and vision-language tasks","author":"Wang","year":"2023"},{"key":"10.1016\/j.neucom.2026.134131_bib0010","doi-asserted-by":"crossref","first-page":"34892","DOI":"10.52202\/075280-1516","article-title":"Visual instruction tuning","volume":"36","author":"Liu","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2026.134131_bib0015","author":"Zhao"},{"key":"10.1016\/j.neucom.2026.134131_bib0020","author":"Bai"},{"key":"10.1016\/j.neucom.2026.134131_bib0025","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"24185","article-title":"InternVL: scaling up vision foundation models and aligning for generic visual-linguistic tasks","author":"Chen","year":"2024"},{"key":"10.1016\/j.neucom.2026.134131_bib0030","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"27218","article-title":"VTQA: visual text question answering via entity alignment and cross-media reasoning","author":"Chen","year":"2024"},{"key":"10.1016\/j.neucom.2026.134131_bib0035","series-title":"Improved Baselines with Visual Instruction Tuning","author":"Liu","year":"2023"},{"key":"10.1016\/j.neucom.2026.134131_bib0040","first-page":"611","article-title":"An image is worth 16x16 words: transformers for image recognition at scale","author":"Dosovitskiy","year":"2021","journal-title":"ICLR"},{"key":"10.1016\/j.neucom.2026.134131_bib0045","author":"Child"},{"key":"10.1016\/j.neucom.2026.134131_bib0050","author":"Wang"},{"key":"10.1016\/j.neucom.2026.134131_bib0055","author":"Kitaev"},{"key":"10.1016\/j.neucom.2026.134131_bib0060","series-title":"Computer Vision - ECCV 2014 - 13th European Conference, Zurich, Switzerland, September 6\u201312, 2014, Proceedings, Part V, Vol. 8693 of Lecture Notes in Computer Science","first-page":"740","article-title":"Microsoft COCO: common objects in context","author":"Lin","year":"2014"},{"key":"10.1016\/j.neucom.2026.134131_bib0065","series-title":"2015 IEEE International Conference on Computer Vision (ICCV)","first-page":"2425","article-title":"VQA: visual question answering","author":"Antol","year":"2015"},{"key":"10.1016\/j.neucom.2026.134131_bib0070","series-title":"2017 IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2017","first-page":"6325","article-title":"Making the V in VQA matter: elevating the role of image understanding in visual question answering","author":"Goyal","year":"2017"},{"key":"10.1016\/j.neucom.2026.134131_bib0075","series-title":"2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"4995","article-title":"Visual7W: grounded question answering in images","author":"Zhu","year":"2016"},{"issue":"10","key":"10.1016\/j.neucom.2026.134131_bib0080","doi-asserted-by":"crossref","first-page":"2413","DOI":"10.1109\/TPAMI.2017.2754246","article-title":"FVQA: fact-based visual question answering","volume":"40","author":"Wang","year":"2018","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.neucom.2026.134131_bib0085","doi-asserted-by":"crossref","first-page":"2507","DOI":"10.52202\/068431-0182","article-title":"Learn to explain: multimodal reasoning via thought chains for science question answering","volume":"35","author":"Lu","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2026.134131_bib0090","series-title":"2017 IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2017","first-page":"5376","article-title":"Are you smarter than a sixth grader? Textbook question answering for multimodal machine comprehension","author":"Kembhavi","year":"2017"},{"key":"10.1016\/j.neucom.2026.134131_bib0095","series-title":"MultiModalQA: Complex Question Answering Over Text, Tables and Images","author":"Talmor","year":"2021"},{"key":"10.1016\/j.neucom.2026.134131_bib0100","series-title":"Proceedings of the 31st ACM International Conference on Multimedia, MM \u201923","first-page":"9487","article-title":"Answer-based entity extraction and alignment for visual text question answering","author":"Yu","year":"2023"},{"key":"10.1016\/j.neucom.2026.134131_bib0105","series-title":"Proceedings of the 31st ACM International Conference on Multimedia, MM \u201923","first-page":"9456","article-title":"VTQAGen: BART-based generative model for visual text question answering","author":"Chen","year":"2023"},{"key":"10.1016\/j.neucom.2026.134131_bib0110","series-title":"Proceedings of the 31st ACM International Conference on Multimedia, MM \u201923","first-page":"9420","article-title":"Finetuning language models for multimodal question answering","author":"Zhang","year":"2023"},{"key":"10.1016\/j.neucom.2026.134131_bib0115","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"8317","article-title":"Towards VQA models that can read","author":"Singh","year":"2019"},{"key":"10.1016\/j.neucom.2026.134131_bib0120","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"4291","article-title":"Scene text visual question answering","author":"Biten","year":"2019"},{"key":"10.1016\/j.neucom.2026.134131_bib0125","doi-asserted-by":"crossref","first-page":"35549","DOI":"10.52202\/068431-2576","article-title":"Towards video text visual question answering: benchmark and baseline","volume":"35","author":"Zhao","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2026.134131_bib0130","series-title":"International Conference on Document Analysis and Recognition","first-page":"137","article-title":"Reading between the lanes: text videoqa on the road","author":"Tom","year":"2023"},{"key":"10.1016\/j.neucom.2026.134131_bib0135","series-title":"Proceedings of the Computer Vision and Pattern Recognition Conference","first-page":"3363","article-title":"Egotextvqa: towards egocentric scene-text aware video question answering","author":"Zhou","year":"2025"},{"key":"10.1016\/j.neucom.2026.134131_bib0140","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"16548","article-title":"Latr: layout-aware transformer for scene-text VQA","author":"Biten","year":"2022"},{"issue":"4","key":"10.1016\/j.neucom.2026.134131_bib0145","first-page":"1","article-title":"Graph pooling inference network for text-based VQA","volume":"20","author":"Zhou","year":"2024","journal-title":"ACM Transactions on Multimedia Computing, Communications and Applications"},{"key":"10.1016\/j.neucom.2026.134131_bib0150","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2025.126478","article-title":"Lightweight Text-VQA: superior trade-off between performance and capacity","volume":"270","author":"Bi","year":"2025","journal-title":"Expert Syst. Appl."},{"key":"10.1016\/j.neucom.2026.134131_bib0155","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence, 39","first-page":"10275","article-title":"Track the answer: extending textvqa from image to video with spatio-temporal clues","author":"Zhang","year":"2025"},{"key":"10.1016\/j.neucom.2026.134131_bib0160","series-title":"Proceedings of the 33rd ACM International Conference on Multimedia","first-page":"876","article-title":"Gather and trace: rethinking video textvqa from an instance-oriented perspective","author":"Zhang","year":"2025"},{"key":"10.1016\/j.neucom.2026.134131_bib0165","doi-asserted-by":"crossref","first-page":"1417","DOI":"10.1109\/TMM.2025.3639956","article-title":"Scene-text grounding for text-based video question answering","volume":"28","author":"Zhou","year":"2026","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.neucom.2026.134131_bib0170","author":"Wang"},{"key":"10.1016\/j.neucom.2026.134131_bib0175","author":"Zhu"},{"key":"10.1016\/j.neucom.2026.134131_bib0180","article-title":"LLaVA-Video: video instruction tuning with synthetic data","author":"Zhang","year":"2025","journal-title":"Trans. Mach. Learn. Res."},{"key":"10.1016\/j.neucom.2026.134131_bib0185","first-page":"19941","article-title":"Multimodal large language models for text-rich image understanding: a comprehensive review","author":"Fu","year":"2025","journal-title":"Findings of the Association for Computational Linguistics: ACL 2025"},{"key":"10.1016\/j.neucom.2026.134131_bib0190","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"26763","article-title":"Monkey: image resolution and text label are important things for large multi-modal models","author":"Li","year":"2024"},{"issue":"5","key":"10.1016\/j.neucom.2026.134131_bib0195","doi-asserted-by":"crossref","first-page":"6008","DOI":"10.1109\/TPAMI.2026.3653415","article-title":"Textmonkey: an OCR-free large multimodal model for understanding document","volume":"48","author":"Liu","year":"2026","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.neucom.2026.134131_bib0200","author":"Zeng"},{"key":"10.1016\/j.neucom.2026.134131_bib0205","series-title":"Proceedings of the Computer Vision and Pattern Recognition Conference","first-page":"9740","article-title":"Token cropr: faster vits for quite a few tasks","author":"Bergner","year":"2025"},{"key":"10.1016\/j.neucom.2026.134131_bib0210","author":"Jeddi"},{"key":"10.1016\/j.neucom.2026.134131_bib0215","series-title":"AdaptVision: dynamic input scaling in MLLMs for versatile scene understanding","author":"Wang","year":"2024"},{"key":"10.1016\/j.neucom.2026.134131_bib0220","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"13796","article-title":"Regiongpt: towards region understanding vision language model","author":"Guo","year":"2024"},{"key":"10.1016\/j.neucom.2026.134131_bib0225","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV) Workshops","first-page":"7543","article-title":"Describe anything model for visual question answering on text-rich images","author":"Vu","year":"2025"},{"key":"10.1016\/j.neucom.2026.134131_bib0230","author":"Zhu"},{"key":"10.1016\/j.neucom.2026.134131_bib0235","author":"Xia"},{"key":"10.1016\/j.neucom.2026.134131_bib0240","series-title":"2023 3rd Asia-Pacific Conference on Communications Technology and Computer Science (ACCTCS)","first-page":"552","article-title":"A study of visual question answering techniques based on collaborative multi-head attention","author":"Yang","year":"2023"},{"issue":"7","key":"10.1016\/j.neucom.2026.134131_bib0245","doi-asserted-by":"crossref","first-page":"544","DOI":"10.1007\/s10489-025-06325-4","article-title":"Multi-scale dual-stream visual feature extraction and graph reasoning for visual question answering","volume":"55","author":"Yusuf","year":"2025","journal-title":"Appl. Intell."},{"key":"10.1016\/j.neucom.2026.134131_bib0250","series-title":"Proceedings of the 25th ACM International Conference on Multimedia","first-page":"1645","article-title":"Video question answering via gradually refined attention over appearance and motion","author":"Xu","year":"2017"},{"key":"10.1016\/j.neucom.2026.134131_bib0255","series-title":"Proceedings of the Thirty-First AAAI Conference on Artificial Intelligence, 31","first-page":"4334","article-title":"Leveraging video descriptions to learn video question answering","author":"Zeng","year":"2017"},{"key":"10.1016\/j.neucom.2026.134131_bib0260","series-title":"Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing","first-page":"1369","article-title":"Tvqa: localized, compositional video question answering","author":"Lei","year":"2018"},{"key":"10.1016\/j.neucom.2026.134131_bib0265","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"9777","article-title":"Next-qa: next phase of question-answering to explaining temporal actions","author":"Xiao","year":"2021"},{"key":"10.1016\/j.neucom.2026.134131_bib0270","doi-asserted-by":"crossref","first-page":"76749","DOI":"10.52202\/075280-3354","article-title":"Self-chained image-language model for video localization and question answering","volume":"36","author":"Yu","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2026.134131_bib0275","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"15405","article-title":"Hitea: hierarchical temporal-aware video-language pre-training","author":"Ye","year":"2023"},{"key":"10.1016\/j.neucom.2026.134131_bib0280","first-page":"1","article-title":"Language-guided progressive attention for visual grounding in remote sensing images","volume":"62","author":"Li","year":"2024","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"10.1016\/j.neucom.2026.134131_bib0285","series-title":"Proceedings of the 38th International Conference on Machine Learning, Vol. 139 of Proceedings of Machine Learning Research","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"issue":"140","key":"10.1016\/j.neucom.2026.134131_bib0290","first-page":"1","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","volume":"21","author":"Raffel","year":"2020","journal-title":"J. Mach. Learn. Res."},{"key":"10.1016\/j.neucom.2026.134131_bib0295","author":"Achiam"},{"key":"10.1016\/j.neucom.2026.134131_bib0300","series-title":"Advances in Neural Information Processing Systems, 35","first-page":"32897","article-title":"VLMo: unified vision-language pre-training with mixture-of-modality-experts","author":"Bao","year":"2022"},{"key":"10.1016\/j.neucom.2026.134131_bib0305","doi-asserted-by":"crossref","first-page":"49250","DOI":"10.52202\/075280-2142","article-title":"InstructBLIP: towards general-purpose vision-language models with instruction tuning","volume":"36","author":"Dai","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2026.134131_bib0310","author":"Wang"},{"key":"10.1016\/j.neucom.2026.134131_bib0315","series-title":"International Conference on Machine Learning","first-page":"19730","article-title":"Blip-2: bootstrapping language-image pre-training with frozen image encoders and large language models","author":"Li","year":"2023"}],"container-title":["Neurocomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0925231226015298?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0925231226015298?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,8,7]],"date-time":"2026-08-07T23:46:48Z","timestamp":1786146408000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0925231226015298"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,10]]},"references-count":63,"alternative-id":["S0925231226015298"],"URL":"https:\/\/doi.org\/10.1016\/j.neucom.2026.134131","relation":{},"ISSN":["0925-2312"],"issn-type":[{"value":"0925-2312","type":"print"}],"subject":[],"published":{"date-parts":[[2026,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Looking closer and smarter: Multi-scale progressive attention for visual text question answering","name":"articletitle","label":"Article Title"},{"value":"Neurocomputing","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.neucom.2026.134131","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Published by Elsevier B.V.","name":"copyright","label":"Copyright"}],"article-number":"134131"}}