{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,7,4]],"date-time":"2025-07-04T05:38:19Z","timestamp":1751607499978,"version":"3.37.3"},"reference-count":32,"publisher":"IEEE","license":[{"start":{"date-parts":[[2021,10,23]],"date-time":"2021-10-23T00:00:00Z","timestamp":1634947200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2021,10,23]],"date-time":"2021-10-23T00:00:00Z","timestamp":1634947200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2021,10,23]],"date-time":"2021-10-23T00:00:00Z","timestamp":1634947200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61771197"],"award-info":[{"award-number":["61771197"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021,10,23]]},"DOI":"10.1109\/cisp-bmei53629.2021.9624447","type":"proceedings-article","created":{"date-parts":[[2021,12,7]],"date-time":"2021-12-07T20:49:58Z","timestamp":1638910198000},"page":"1-5","source":"Crossref","is-referenced-by-count":1,"title":["Visual Question Answering Based on Position Alignment"],"prefix":"10.1109","author":[{"given":"Qihao","family":"Xia","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chao","family":"Yu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Pingping","family":"Peng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Henghao","family":"Gu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhengqi","family":"Zheng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kun","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.202"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N18-2074"},{"key":"ref30","article-title":"Identity Mappings in Deep Residual Networks","author":"he","year":"0","journal-title":"European Conference on Computer Vision"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2016.2577031"},{"key":"ref11","first-page":"807","article-title":"Rectified Linear Units Improve Restricted Boltzmann Machines","author":"nair","year":"0","journal-title":"27th International Conference on Machine Learning"},{"key":"ref12","article-title":"Tips and Tricks for Visual Question Answering: Learnings from the 2017 Challenge","author":"teney","year":"2017","journal-title":"ArXiv Preprint"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.670"},{"key":"ref14","first-page":"1929","article-title":"Dropout: A Simple Way to Prevent Neural Networks from Overfitting","volume":"15","author":"nitish","year":"2014","journal-title":"Journal of Machine Learning Research"},{"key":"ref15","article-title":"Adam: A Method for Stochastic Optimization","author":"kingma","year":"2014","journal-title":"ArXiv Preprint"},{"key":"ref16","article-title":"Bilinear Attention Networks","author":"kim","year":"2018","journal-title":"ArXiv"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D16-1044"},{"key":"ref18","article-title":"Hadamard Product for Low-rank Bilinear Pooling","author":"kim","year":"0","journal-title":"International Conference on Learning Representations"},{"key":"ref19","article-title":"Learning Conditioned Graph Structures for Interpretable Visual Question Answering","author":"norcliffe-brown","year":"2018","journal-title":"ArXiv"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.41"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01068"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.01039"},{"key":"ref3","article-title":"Visual dialog","author":"abhishek","year":"0","journal-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00636"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref5","first-page":"5998","article-title":"Attention is all you need","author":"ashish","year":"0","journal-title":"Advances in neural information processing systems"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/D14-1162"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/D14-1162"},{"key":"ref2","first-page":"2425","article-title":"Vqa: Visual question answering","author":"stanislaw","year":"0","journal-title":"IEEE International Conference on Computer Vision"},{"key":"ref9","first-page":"1724","article-title":"Learning Phrase Representations using RNN EncoderDecoder for Statistical Machine Translation","author":"cho","year":"0","journal-title":"2014 Conference on Empirical Methods in Natural Language Processing"},{"key":"ref1","first-page":"2048","article-title":"Show, attend and tell: neural image caption generation with visual attention","author":"xu","year":"0","journal-title":"International Conference on Machine Learning PMLR 37"},{"key":"ref20","article-title":"Learning to Count Objects in Natural Images for Visual Question Answering","author":"zhang","year":"2018","journal-title":"ArXiv"},{"key":"ref22","article-title":"Bilinear classifiers for visual recognition","author":"pirsiavash","year":"0","journal-title":"NIPS"},{"key":"ref21","article-title":"Dynamic Fusion With Intra- and Inter-Modality Attention Flow for Visual Question Answering","author":"peng","year":"0","journal-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)"},{"key":"ref24","first-page":"2008","article-title":"Spatial Transformer Networks","volume":"28","author":"jaderberg","year":"0","journal-title":"Advances in neural information processing systems"},{"key":"ref23","article-title":"Visual genome: Connecting language and vision using crowdsourced dense image annotations","author":"krishna","year":"2016","journal-title":"ArXiv Preprint"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2021.3104937"},{"key":"ref25","article-title":"A Simple Loss Function for Improving the Convergence and Accuracy of Visual Question Answering Models","author":"ilievski","year":"2017","journal-title":"Computer Vision and Pattern Recognition"}],"event":{"name":"2021 14th International Congress on Image and Signal Processing, BioMedical Engineering and Informatics (CISP-BMEI)","start":{"date-parts":[[2021,10,23]]},"location":"Shanghai, China","end":{"date-parts":[[2021,10,25]]}},"container-title":["2021 14th International Congress on Image and Signal Processing, BioMedical Engineering and Informatics (CISP-BMEI)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9624201\/9624206\/09624447.pdf?arnumber=9624447","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,5,10]],"date-time":"2022-05-10T16:53:40Z","timestamp":1652201620000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9624447\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,10,23]]},"references-count":32,"URL":"https:\/\/doi.org\/10.1109\/cisp-bmei53629.2021.9624447","relation":{},"subject":[],"published":{"date-parts":[[2021,10,23]]}}}