{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,1]],"date-time":"2025-11-01T09:36:11Z","timestamp":1761989771068,"version":"3.28.0"},"reference-count":28,"publisher":"IEEE","license":[{"start":{"date-parts":[[2024,4,14]],"date-time":"2024-04-14T00:00:00Z","timestamp":1713052800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,4,14]],"date-time":"2024-04-14T00:00:00Z","timestamp":1713052800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,4,14]]},"DOI":"10.1109\/icassp48485.2024.10445873","type":"proceedings-article","created":{"date-parts":[[2024,3,18]],"date-time":"2024-03-18T18:56:31Z","timestamp":1710788191000},"page":"8130-8134","source":"Crossref","is-referenced-by-count":1,"title":["Segment then Match: Find the Carrier before Reasoning in Scene-Text VQA"],"prefix":"10.1109","author":[{"given":"Chengyang","family":"Fang","sequence":"first","affiliation":[{"name":"Chinese Academy of Sciences,Institute of Information Engineering"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Liang","family":"Li","sequence":"additional","affiliation":[{"name":"Chinese Academy of Sciences,Institute of Information Engineering"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiapeng","family":"Liu","sequence":"additional","affiliation":[{"name":"Chinese Academy of Sciences,Institute of Information Engineering"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bing","family":"Li","sequence":"additional","affiliation":[{"name":"Chinese Academy of Sciences,Institute of Information Engineering"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dayong","family":"Hu","sequence":"additional","affiliation":[{"name":"Heilongjiang Network Space Research Center"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Can","family":"Ma","sequence":"additional","affiliation":[{"name":"Chinese Academy of Sciences,Institute of Information Engineering"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413924"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00869"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01014"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.coling-main.278"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01001"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ICME52920.2022.9859603"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01605"},{"article-title":"Tag: Boosting text-vqa via text-aware visual question-answer generation","volume-title":"BMVC","author":"Wang","key":"ref8"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW54120.2021.00297"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00864"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475606"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547977"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i9.26357"},{"issue":"1","key":"ref14","first-page":"5485","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","volume-title":"JMLR","volume":"21","author":"Raffel"},{"key":"ref15","first-page":"91","article-title":"Faster R-CNN: towards real-time object detection with region proposal networks","author":"Ren","year":"2015","journal-title":"NeurIPS"},{"key":"ref16","first-page":"715","article-title":"Spatially aware multi-modal transformers for textvqa","volume-title":"ECCV","author":"Kant"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/tpami.2021.3132034"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00851"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-020-01316-z"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00439"},{"journal-title":"Coco-text: Dataset and benchmark for text detection and recognition in natural images","year":"2016","author":"Veit","key":"ref21"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-016-0981-7"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1145\/1805986.1806020"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2015.7333942"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2013.221"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2013.378"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3611753"}],"event":{"name":"ICASSP 2024 - 2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","start":{"date-parts":[[2024,4,14]]},"location":"Seoul, Korea, Republic of","end":{"date-parts":[[2024,4,19]]}},"container-title":["ICASSP 2024 - 2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/10445798\/10445803\/10445873.pdf?arnumber=10445873","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,8,2]],"date-time":"2024-08-02T04:57:40Z","timestamp":1722574660000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10445873\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,4,14]]},"references-count":28,"URL":"https:\/\/doi.org\/10.1109\/icassp48485.2024.10445873","relation":{},"subject":[],"published":{"date-parts":[[2024,4,14]]}}}