{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T16:18:20Z","timestamp":1783095500180,"version":"3.54.6"},"reference-count":109,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2021,1,25]],"date-time":"2021-01-25T00:00:00Z","timestamp":1611532800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,1,25]],"date-time":"2021-01-25T00:00:00Z","timestamp":1611532800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100001809","name":"The National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["No.61702140"],"award-info":[{"award-number":["No.61702140"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"name":"The Scientific Research Foundation for The Overseas Returning Person of Heilongjiang Province of China","award":["LC2018030"],"award-info":[{"award-number":["LC2018030"]}]},{"name":"The Fundamental Research Foundation for Universities of Heilongjiang Province","award":["JMRH2018XM04"],"award-info":[{"award-number":["JMRH2018XM04"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Mobile Netw Appl"],"published-print":{"date-parts":[[2021,10]]},"DOI":"10.1007\/s11036-020-01730-0","type":"journal-article","created":{"date-parts":[[2021,1,25]],"date-time":"2021-01-25T15:05:04Z","timestamp":1611587104000},"page":"1904-1937","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":17,"title":["Video Question Answering: a Survey of Models and Datasets"],"prefix":"10.1007","volume":"26","author":[{"given":"Guanglu","family":"Sun","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lili","family":"Liang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tianlin","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bo","family":"Yu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Meng","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bolun","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2021,1,25]]},"reference":[{"key":"1730_CR1","doi-asserted-by":"crossref","unstructured":"Lin T Y, Doll\u00e1r P, Girshick R, He K, Hariharan B, Belongie S (2017) Feature pyramid networks for object detection[C]. Proceedings of the IEEE conference on computer vision and pattern recognition, 2117\u20132125","DOI":"10.1109\/CVPR.2017.106"},{"key":"1730_CR2","doi-asserted-by":"crossref","unstructured":"Maninis K K , Caelles S , Chen Y (2018) Video object segmentation without temporal information[J]. IEEE Transactions on Pattern Analysis and Machine Intelligence 41(6):1515\u20131530","DOI":"10.1109\/TPAMI.2018.2838670"},{"key":"1730_CR3","doi-asserted-by":"crossref","unstructured":"Wong W K , Lai Z, Wen J, Fang X, Lu Y (2017) Low-rank embedding for robust image feature extraction[J]. IEEE Transactions on Image Processing\u00a026(6): 2905\u20132917","DOI":"10.1109\/TIP.2017.2691543"},{"key":"1730_CR4","doi-asserted-by":"crossref","unstructured":"Lu J, Peng Y, Qi GJ, Jun Y (2020) Guest editorial introduction to the special section on representation learning for visual content understanding[J]. IEEE Transactions on Circuits and Systems for Video Technology\u00a030(9):2797\u20132800","DOI":"10.1109\/TCSVT.2020.3009095"},{"key":"1730_CR5","doi-asserted-by":"crossref","unstructured":"Anjum A, Abdullah T, Tariq MF, Baltaci Y, Antonopoulos N (2019) Video stream analysis in clouds: an object detection and classification framework for high performance video analytics [J]. IEEE Trans Cloud Comput 7(4):1152\u20131167","DOI":"10.1109\/TCC.2016.2517653"},{"key":"1730_CR6","unstructured":"Ren M, Kiros R, Zemel R (2015) Image Question Answering: A Visual Semantic Embedding Model and a New Dataset[J]. Proc. Advances in Neural Inf. Process. Syst 1(2):5."},{"key":"1730_CR7","doi-asserted-by":"crossref","unstructured":"Ren S, He K, Girshick R, Sun J (2016) Faster r-cnn: Towards real-time object detection with region proposal networks[J]. IEEE transactions on pattern analysis and machine intelligence 39(6):1137\u20131149","DOI":"10.1109\/TPAMI.2016.2577031"},{"key":"1730_CR8","unstructured":"Simonyan K, Zisserman A (2014) Very deep convolutional networks for large-scale image recognition[J]. Computer Science"},{"key":"1730_CR9","doi-asserted-by":"crossref","unstructured":"Szegedy C, Vanhoucke V, Ioffe S, Shlens J, Wojna Z (2016) Rethinking the inception architecture for computer vision[C]. Proceedings of the IEEE conference on computer vision and pattern recognition 2818\u20132826","DOI":"10.1109\/CVPR.2016.308"},{"key":"1730_CR10","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition[C]. Proceedings of the IEEE conference on computer vision and pattern recognition 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"1730_CR11","doi-asserted-by":"crossref","unstructured":"Tran D, Bourdev L, Fergus R, Torresani L, Paluri M (2015) Learning spatiotemporal features with 3d convolutional networks[C]. Proceedings of the IEEE international conference on computer vision 4489\u20134497","DOI":"10.1109\/ICCV.2015.510"},{"key":"1730_CR12","unstructured":"Mikolov T, Chen K, Corrado G, Dean J (2013) Efficient estimation of word representations in vector space[J]. Computer Science"},{"key":"1730_CR13","doi-asserted-by":"crossref","unstructured":"Pennington J, Socher R, Manning C (2014) Glove: global vectors for word representation[C]. Proceedings of the 2014 conference on empirical methods in natural language processing 1532\u20131543","DOI":"10.3115\/v1\/D14-1162"},{"key":"1730_CR14","unstructured":"Kiros R, Zhu Y, Salakhutdinov R R, Zemel R, Urtasun R, Torralba A, Fidleret S (2015) Skip-thought vectors[J]. Advances in neural information processing systems 28:3294-3302"},{"key":"1730_CR15","doi-asserted-by":"crossref","unstructured":"Sethy A, Ramabhadran B (2008) Bag-of-word normalized n-gram models[C]. Ninth Annual Conference of the International Speech Communication Association","DOI":"10.21437\/Interspeech.2008-265"},{"key":"1730_CR16","unstructured":"Devlin J, Chang M W, Lee K, Toutanova K (2019) BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding[C]. Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies 4171\u20134186"},{"key":"1730_CR17","doi-asserted-by":"crossref","unstructured":"Gao K, Zhu X, Han Y (2017) Initialized Frame Attention Networks for Video Question Answering[C]. International Conference on Internet Multimedia Computing and Service. Springer, Singapore, pp 349\u2013359","DOI":"10.1007\/978-981-10-8530-7_34"},{"key":"1730_CR18","doi-asserted-by":"crossref","unstructured":"Xu D, Zhao Z, Xiao J, Wu F, Zhang H, He X (2017) Video question answering via gradually refined attention over appearance and motion[C]. Proceedings of the 25th ACM international conference on Multimedia 1645\u20131653","DOI":"10.1145\/3123266.3123427"},{"key":"1730_CR19","doi-asserted-by":"crossref","unstructured":"Tapaswi M, Zhu Y, Stiefelhagen R, Torralba A, Urtasun R, Fidler S (2016) Movieqa: understanding stories in movies through question-answering[C]. Proceedings of the IEEE conference on computer vision and pattern recognition 4631\u20134640","DOI":"10.1109\/CVPR.2016.501"},{"key":"1730_CR20","unstructured":"Yuan Z, Sun S, Duan L, Wu X, Xu C (2019) Adversarial multi-modal network for movie question answering[J].\u00a0arXiv preprint arXiv:1906.09844"},{"key":"1730_CR21","doi-asserted-by":"crossref","unstructured":"Jang Y, Song Y, Yu Y, Kim Y, Kim G (2017) Tgif-qa: toward spatio-temporal reasoning in visual question answering[C]. Proceedings of the IEEE conference on computer vision and pattern recognition 2758\u20132766","DOI":"10.1109\/CVPR.2017.149"},{"key":"1730_CR22","doi-asserted-by":"crossref","unstructured":"Lei J, Yu L, Bansal M, Berg T (2018) Tvqa: localized, compositional video question answering[C]. Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing 1369\u20131379","DOI":"10.18653\/v1\/D18-1167"},{"key":"1730_CR23","doi-asserted-by":"crossref","unstructured":"Kim K M, Heo M O, Choi S H, Zhang B T (2017) Deepstory: Video story qa by deep embedded memory networks[C]. Proceedings of the Twenty-Sixth International Joint Conference on Artificial Intelligence 2016\u20132022","DOI":"10.24963\/ijcai.2017\/280"},{"key":"1730_CR24","doi-asserted-by":"crossref","unstructured":"Yu Z, Xu D, Yu J, Zhao Z, Zhuang Y (2019) ActivityNet-QA: a dataset for understanding complex web videos via question answering[C]. AAAI 2019: thirty-third AAAI conference on artificial intelligence 33(1):9127\u20139134","DOI":"10.1609\/aaai.v33i01.33019127"},{"key":"1730_CR25","unstructured":"Malinowski M, Fritz M (2014) A multi-world approach to question answering about real-world scenes based on uncertain input[J].\u00a0Advances in Neural Information Processing Systems 27:1682\u20131690"},{"key":"1730_CR26","doi-asserted-by":"crossref","unstructured":"Wu Q, Teney D, Wang P, Shen C, Dick A, van den Hengel A (2017) Visual question answering: a survey of methods and datasets[J]. \u00a0Computer Vision and Image Understanding 163:21\u201340","DOI":"10.1016\/j.cviu.2017.05.001"},{"key":"1730_CR27","doi-asserted-by":"crossref","unstructured":"Kafle K, Kanan C (2017) Visual question answering: datasets, algorithms, and future challenges[J]. Comput Vis Image Underst 163:3\u201320","DOI":"10.1016\/j.cviu.2017.06.005"},{"key":"1730_CR28","unstructured":"Gupta A K (2017) Survey of visual question answering: datasets and techniques[J]. arXiv preprint arXiv:1705.03865"},{"key":"1730_CR29","unstructured":"Pandhre S, Sodhani S (2017) Survey of recent advances in visual question answering[J]. arXiv preprint arXiv:1709.08203"},{"key":"1730_CR30","doi-asserted-by":"crossref","unstructured":"Zhang D, Cao R, Wu S (2019) Information fusion in visual question answering: a survey [J]. Information Fusion 52:268\u2013280","DOI":"10.1016\/j.inffus.2019.03.005"},{"key":"1730_CR31","doi-asserted-by":"crossref","unstructured":"Zhu L, Xu Z, Yang Y, Hauptmann AG (2017) Uncovering the temporal context for video question answering[J]. International Journal of Computer Vision 124(3):409\u2013421","DOI":"10.1007\/s11263-017-1033-7"},{"key":"1730_CR32","doi-asserted-by":"crossref","unstructured":"Wang Y S, Su H T, Chang C H, Liu Z, Hsu QW H (2019) Video question generation via cross-modal self-attention networks learning[C]. International Conference on Acoustics, Speech and Signal Processing 2423\u20132427","DOI":"10.1109\/ICASSP40776.2020.9053476"},{"key":"1730_CR33","doi-asserted-by":"crossref","unstructured":"Jin W, Zhao Z, Li Y et al (2019) Video question answering via knowledge-based progressive spatial-temporal attention network[J]. ACM Trans Multimed Comput Commun Appl (TOMM) 15(2):1\u201322","DOI":"10.1145\/3321505"},{"key":"1730_CR34","doi-asserted-by":"crossref","unstructured":"Kim J, Ma M, Kim K, Kim S, Yoo C D (2019) Gaining Extra Supervision via Multi-task learning for Multi-Modal Video Question Answering[C]. 2019 International Joint Conference on Neural Networks (IJCNN) IEEE 1\u20138\u00a0","DOI":"10.1109\/IJCNN.2019.8852087"},{"key":"1730_CR35","unstructured":"Xiao S, Li Y, Ye Y, Zhao Z, Xiao, Wu F, Zhu J, Zhuang Y (2015) Video question answering via multi-granularity temporal attention network learning[C]. Proceedings of the 10th International Conference on Internet Multimedia Computing and Service 1\u20135"},{"key":"1730_CR36","doi-asserted-by":"crossref","unstructured":"Xiao S, Li Y, Ye Y, Chen L, Pu S, Zhao Z, Shao J, Xiao J (2020) Hierarchical Temporal Fusion of Multi-grained Attention Features for Video Question Answering[J].\u00a0Neural Processing Letters 52(2):993\u20131003","DOI":"10.1007\/s11063-019-10003-1"},{"key":"1730_CR37","doi-asserted-by":"crossref","unstructured":"Lei J, Yu L, Berg T L, Bansal M (2020) TVQA+: Spatio-temporal grounding for video question answering[C].\u00a0Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics 8211\u20138225","DOI":"10.18653\/v1\/2020.acl-main.730"},{"key":"1730_CR38","doi-asserted-by":"crossref","unstructured":"Xue H, Zhao Z, Cai D (2017) Unifying the video and question attentions for open-ended video question answering[J]. IEEE Transactions on Image Processing 26(12):5656\u20135666","DOI":"10.1109\/TIP.2017.2746267"},{"key":"1730_CR39","doi-asserted-by":"crossref","unstructured":"Zhao Z, Yang Q, Cai D, He X, Zhuang Y, Zhao Z (2017) Video question answering via hierarchical Spatio-temporal attention networks[C]. twenty-sixth international joint conference on artificial Intelligence 3518\u20133524","DOI":"10.24963\/ijcai.2017\/492"},{"key":"1730_CR40","doi-asserted-by":"crossref","unstructured":"Zhao Z, Zhang Z, Xiao S, Yu Z, Yu J, Cai D, F Wu (2018) Open-Ended Long-form Video Question Answering via Adaptive Hierarchical Reinforced Networks[C]. IJCAI 2018: 27th International Joint Conference on Artificial Intelligence 3683\u20133689","DOI":"10.24963\/ijcai.2018\/512"},{"key":"1730_CR41","doi-asserted-by":"crossref","unstructured":"Zhao Z, Zhang Z, Xiao S, Xiao Z, Yan X, Yu J, Cai D (2019) Long-form video question answering via dynamic hierarchical reinforced networks[J]. IEEE Transactions on Image Processing 28(12):5939\u20135925","DOI":"10.1109\/TIP.2019.2922062"},{"key":"1730_CR42","doi-asserted-by":"crossref","unstructured":"Yu Y, Kim J, Kim G (2018) A joint sequence fusion model for video question answering and retrieval[C]. Proceedings of the European Conference on Computer Vision (ECCV) 471\u2013487","DOI":"10.1007\/978-3-030-01234-2_29"},{"key":"1730_CR43","doi-asserted-by":"crossref","unstructured":"Zhao Z, Lin J, Jiang X, Cai D, He X, Zhuang Y (2017) Video question answering via hierarchical dual-level attention network learning[C]. Proceedings of the 25th ACM International Conference on Multimedia 1050\u20131058\u00a0","DOI":"10.1145\/3123266.3123364"},{"key":"1730_CR44","doi-asserted-by":"crossref","unstructured":"Xue H, Chu W, Zhao Z, Cai D (2018) A better way to attend: attention with trees for video question answering[J]. IEEE Transactions on Image Processing 27(11):5563\u20135574","DOI":"10.1109\/TIP.2018.2859820"},{"key":"1730_CR45","doi-asserted-by":"crossref","unstructured":"Zhao Z, Jiang X, Cai, XJ, He X, Pu S (2018)\u00a0Multi-Turn Video Question Answering via Multi-Stream Hierarchical Attention Context Network[C]. IJCAI 2018: 27th International Joint Conference on Artificial Intelligence, 3690\u20133696\u00a0","DOI":"10.24963\/ijcai.2018\/513"},{"key":"1730_CR46","doi-asserted-by":"crossref","unstructured":"Zhao Z, Zhang Z, Jiang X, Cai D (2019) Multi-turn video question answering via hierarchical attention context reinforced networks[J]. IEEE Transactions on Image Processing 28(8):3860\u20133872","DOI":"10.1109\/TIP.2019.2902106"},{"key":"1730_CR47","doi-asserted-by":"crossref","unstructured":"Chu W, Xue H, Zhao Z, Cai D, Yao C (2018) The forgettable-watcher model for video question answering[J]. Neurocomputing 314:386\u2013393","DOI":"10.1016\/j.neucom.2018.06.069"},{"key":"1730_CR48","doi-asserted-by":"crossref","unstructured":"Gao F, Ge Y, Liu Y (2018) Remember and forget: video and text fusion for video question answering [J]. Multimedia Tools and Applications 77(22):29269\u201329282","DOI":"10.1007\/s11042-018-5868-x"},{"key":"1730_CR49","doi-asserted-by":"crossref","unstructured":"Fan C, Zhang X, Zhang S, Wang W, Zhang C, and Huang H \u00a0(2019) Heterogeneous memory enhanced multi-modal attention model for video question answering [C]. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition 1999\u20132007","DOI":"10.1109\/CVPR.2019.00210"},{"key":"1730_CR50","doi-asserted-by":"crossref","unstructured":"Zeng K H, Chen T H, Chuang C Y, Liao Y H, Niebles J C, Sun M (2017) Leveraging video descriptions to learn video question answering[C]. Proceedings of the AAAI Conference on Artificial Intelligence 31(1)","DOI":"10.1609\/aaai.v31i1.11238"},{"key":"1730_CR51","doi-asserted-by":"crossref","unstructured":"Wang B, Xu Y, Han Y, Hong R \u00a0(2018) Movie question answering: remembering the textual cues for layered visual contents[C]. Proceedings of the AAAI Conference on Artificial Intelligence 32(1)","DOI":"10.1609\/aaai.v32i1.12253"},{"key":"1730_CR52","doi-asserted-by":"crossref","unstructured":"Han Y, Wang B, Hong R, Wu F\u00a0(2019)\u00a0Movie question answering via textual memory and plot graph[J]. IEEE Transactions on Circuits and Systems for Video Technology 30(3):875\u2013887","DOI":"10.1109\/TCSVT.2019.2897604"},{"key":"1730_CR53","doi-asserted-by":"crossref","unstructured":"Maharaj T, Ballas N, Rohrbach A, Courville A, Pal C (2017) A dataset and exploration of models for understanding video data through fill-in-the-blank question-answering[C]. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition 7359\u20137368","DOI":"10.1109\/CVPR.2017.778"},{"key":"1730_CR54","doi-asserted-by":"crossref","unstructured":"Ge Y, Xu Y, Han Y (2017) Video question answering using a forget memory network[C]. CCF Chinese Conference on Computer Vision 404\u2013415","DOI":"10.1007\/978-981-10-7299-4_33"},{"key":"1730_CR55","doi-asserted-by":"crossref","unstructured":"Kim J, Ma M, Kim K, Kim S, Yoo C D\u00a0(2020)\u00a0Progressive attention memory network for movie story question answering[C], Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition 8337\u20138346","DOI":"10.1109\/CVPR.2019.00853"},{"key":"1730_CR56","doi-asserted-by":"crossref","unstructured":"Ye Y, Zhao Z, Li Y, Chen L (2017)\u00a0Video question answering via attribute-augmented attention network learning[C]. Proceedings of the 40th International ACM SIGIR Conference on Research and Development in Information Retrieval 829\u2013832","DOI":"10.1145\/3077136.3080655"},{"key":"1730_CR57","doi-asserted-by":"crossref","unstructured":"Song X, Shi Y, Chen X, Han Y (2018) Explore multi-step reasoning in video question answering[C]. Proceedings of the 26th ACM International Conference on Multimedia 239\u2013247","DOI":"10.1145\/3240508.3240563"},{"key":"1730_CR58","unstructured":"Le T M, Le V, Venkatesh S.Tran (2019)\u00a0Learning to Reason with Relational Video Representation for Question Answering[J]. CoRR, abs\/1907.04553"},{"key":"1730_CR59","doi-asserted-by":"crossref","unstructured":"Yu Y, Ko H, Choi J, Kim G (2017)\u00a0End-to-end concept word detection for video captioning, retrieval, and question answering[C]. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition 3261\u20133269","DOI":"10.1109\/CVPR.2017.347"},{"key":"1730_CR60","doi-asserted-by":"crossref","unstructured":"Li X, Song J, Gao L, Liu X, Huang W, He X\u00a0(2019)\u00a0Beyond rnns: positional self-attention with co-attention for video question answering[C]. Proceedings of the AAAI Conference on Artificial Intelligence 33:8658\u20138665","DOI":"10.1609\/aaai.v33i01.33018658"},{"key":"1730_CR61","doi-asserted-by":"crossref","unstructured":"Kim K M, Choi S H, Kim J H, Zhang B T\u00a0(2018)\u00a0Multi-modal dual attention memory for video story question answering[C]. Proceedings of the European Conference on Computer vision 673\u2013688","DOI":"10.1007\/978-3-030-01267-0_41"},{"key":"1730_CR62","doi-asserted-by":"crossref","unstructured":"Na S, Lee S, Kim J, Kim G A\u00a0(2017)\u00a0read-write memory network for movie story understanding[C]. Proceedings of the IEEE International Conference on Computer Vision 677\u2013685","DOI":"10.1109\/ICCV.2017.80"},{"key":"1730_CR63","doi-asserted-by":"crossref","unstructured":"Gao J, Ge R, Chen K, Nevatia R\u00a0(2018)\u00a0Motion-appearance co-memory networks for video question answering[C]. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition 6576\u20136585","DOI":"10.1109\/CVPR.2018.00688"},{"key":"1730_CR64","doi-asserted-by":"crossref","unstructured":"Zhang Z, Zhao Z, Lin Z, Song J, He X\u00a0(2019)\u00a0Open-ended long-form video question answering via hierarchical convolutional self-attention networks[J]. Proceedings of the Twenty-Eighth International Joint Conference on Artificial Intelligence 4383\u20134389","DOI":"10.24963\/ijcai.2019\/609"},{"key":"1730_CR65","doi-asserted-by":"crossref","unstructured":"Mun J, Hongsuck Seo P, Jung I, Han B\u00a0(2017)\u00a0Marioqa: answering questions by watching gameplay videos[C]. Proceedings of the IEEE International Conference on Computer Vision 2867\u20132875","DOI":"10.1109\/ICCV.2017.312"},{"key":"1730_CR66","unstructured":"Radford A, Narasimhan K, Salimans T, Sutskever I (2018) Improving language understanding with unsupervised learning[J]. Technical report, OpenAI"},{"key":"1730_CR67","doi-asserted-by":"crossref","unstructured":"Peters M E, Neumann M, Iyyer M, Gardner M, Clark C, Lee K, Zettlemoyer L (2018) Deep contextualized word representations[J].\u00a0Proceedings of the 2018 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies\u00a01:2227\u20132237","DOI":"10.18653\/v1\/N18-1202"},{"key":"1730_CR68","unstructured":"Mikolov T, Sutskever I, Chen K, Corrado GS (2013) Distributed representations of words and phrases and their compositionality[C]. Advances in neural information processing systems 26:3111\u20133119"},{"key":"1730_CR69","doi-asserted-by":"crossref","unstructured":"Chao G L, Rastogi A, Yavuz S, Hakkani-Tur T, Chen J, Lane I (2019) Learning question-guided video representation for multi-turn video question answering[J]. Proceedings of the 20th Annual SIGdial Meeting on Discourse and Dialogue 215\u2013225","DOI":"10.18653\/v1\/W19-5926"},{"key":"1730_CR70","doi-asserted-by":"crossref","unstructured":"Fukui A, Park D H, Yang D, Rohrbach A (2016) Multi-modal compact bilinear pooling for visual question answering and visual grounding[J].\u00a0arXiv preprint arXiv:1606.01847","DOI":"10.18653\/v1\/D16-1044"},{"key":"1730_CR71","doi-asserted-by":"crossref","unstructured":"Hochreiter S, Schmidhuber J\u00a0(1997)\u00a0Long short-term memory[J]. Neural computation 9(8):1735\u20131780","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"1730_CR72","doi-asserted-by":"crossref","unstructured":"Schuster M, Paliwal KK (1997) Bidirectional recurrent neural networks[J]. IEEE transactions on Signal Processing 45(11):2673\u20132681","DOI":"10.1109\/78.650093"},{"key":"1730_CR73","doi-asserted-by":"crossref","unstructured":"Xu Y, Wang L, Cheng J, Xia H, Yin J (2017) DTA: double LSTM with temporal-wise attention network for action recognition [C]. 2017 3rd IEEE International Conference on Computer and Communications 1676\u20131680","DOI":"10.1109\/CompComm.2017.8322825"},{"key":"1730_CR74","unstructured":"Chung J, Gulcehre C, Cho K H, Bengio Y (2014) Empirical evaluation of gated recurrent neural networks on sequence modeling[J]. CoRR,\u00a0abs\/1412.3555"},{"issue":"6","key":"1730_CR75","first-page":"062007","volume":"322","author":"Z Liujie","year":"2018","unstructured":"Liujie Z, Yanquan Z, Xiuyu D, Ruiqi C (2018) A Hierarchical multi-input and output Bi-GRU Model for Sentiment Analysis on Customer Reviews[J]. IOP Conf Ser: Mater Sci Eng 322(6):062007","journal-title":"IOP Conf Ser: Mater Sci Eng"},{"key":"1730_CR76","doi-asserted-by":"crossref","unstructured":"Yu Z, Yu J, Fan J, Tao D (2017) Multi-modal factorized bilinear pooling with co-attention learning for visual question answering[C]. Proceedings of the IEEE International Conference on Computer Vision 1839\u20131848","DOI":"10.1109\/ICCV.2017.202"},{"key":"1730_CR77","unstructured":"Kim J H, On K W, Lim W, Kim J, Ha JW (2017) Hadamard product for low-rank bilinear pooling[C]. 5th International Conference on Learning Representations"},{"key":"1730_CR78","unstructured":"Sutskever I, Vinyals O, Le QV (2014) Sequence to sequence learning with neural networks[C]. Advances in Neural Information Processing Systems 27:3104\u20133112"},{"key":"1730_CR79","doi-asserted-by":"crossref","unstructured":"Cho K, Van Merri\u00ebnboer B, Bahdanau D, Bengio Y (2014) On the properties of neural machine translation: encoder-decoder approaches[J]. Proceedings of SSST-8, Eighth Workshop on Syntax, Semantics and Structure in Statistical Translation 103\u2013111","DOI":"10.3115\/v1\/W14-4012"},{"key":"1730_CR80","doi-asserted-by":"crossref","unstructured":"Cho K, Van Merri\u00ebnboer B, Gulcehre C, Bengio Y (2014) Learning phrase representations using RNN encoder-decoder for statistical machine translation[C]. Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing 1724\u20131734","DOI":"10.3115\/v1\/D14-1179"},{"key":"1730_CR81","unstructured":"Wang F, Tax D M J (2016) Survey on the attention based RNN model and its applications in computer vision[J]. CoRR,\u00a0abs\/1601.06823"},{"key":"1730_CR82","unstructured":"Seo M, Kembhavi A, Farhadi A, Hajishirzi H (2017) Bidirectional attention flow for machine comprehension[C].\u00a05th International Conference on Learning Representations"},{"key":"1730_CR83","unstructured":"Bahdanau D, Cho K, Bengio Y\u00a0(2015)\u00a0Neural machine translation by jointly learning to align and translate[J]. 3rd International Conference on Learning Representations"},{"key":"1730_CR84","doi-asserted-by":"crossref","unstructured":"Luong M T, Pham H, Manning C D\u00a0(2015)\u00a0Effective approaches to attention-based neural machine translation[J]. Proceedings of the 2015 Conference on Empirical Methods in Natural Language Processing 1412\u20131421","DOI":"10.18653\/v1\/D15-1166"},{"key":"1730_CR85","unstructured":"Larochelle H, Hinton GE (2010) Learning to combine foveal glimpses with a third-order Boltzmann machine[C]. Advances in Neural Information Processing Systems 23:1243\u20131251"},{"key":"1730_CR86","doi-asserted-by":"crossref","unstructured":"Fu J, Zheng H, Mei T\u00a0(2017)\u00a0Look closer to see better: recurrent attention convolutional neural network for fine-grained image recognition[C]. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition 4438\u20134446","DOI":"10.1109\/CVPR.2017.476"},{"key":"1730_CR87","unstructured":"Vaswani A, Shazeer N, Parmar N, Jones L, Gomez A N, Kaiser \u0141, Polosukhin I (2017)\u00a0Attention is all you need[C]. Advances in Neural Information Processing Systems 5998\u20136008"},{"key":"1730_CR88","unstructured":"Chaudhari S, Polatkan G, Ramanath R, et al (2019)\u00a0An attentive survey of attention models[J]. CoRR, abs\/1904.02874"},{"key":"1730_CR89","doi-asserted-by":"crossref","unstructured":"Hu D (2019) An introductory survey on attention mechanisms in NLP problems[C]. Proceedings of SAI Intelligent Systems Conference 432\u2013448","DOI":"10.1007\/978-3-030-29513-4_31"},{"key":"1730_CR90","doi-asserted-by":"crossref","unstructured":"Duval E (2011) Attention please! Learning analytics for visualization and recommendation[C]. Proceedings of the 1st International Conference on Learning Analytics and Knowledge 9\u201317","DOI":"10.1145\/2090116.2090118"},{"key":"1730_CR91","unstructured":"Graves A, Wayne G, Danihelka I(2014) Neural turing machines[J]. arXiv preprint arXiv:1410.5401"},{"key":"1730_CR92","doi-asserted-by":"crossref","unstructured":"Auer S, Bizer C, Kobilarov G, Lehmann J, Cyganiak R, Ives Z (2007) Dbpedia: A nucleus for a web of open data[M], The semantic web. Springer, Berlin, Heidelberg, pp 722\u2013735","DOI":"10.1007\/978-3-540-76298-0_52"},{"key":"1730_CR93","unstructured":"Le Q, Mikolov T (2014) Distributed representations of sentences and documents[C]. International Conference on Machine Learning 1188\u20131196\u00a0"},{"key":"1730_CR94","doi-asserted-by":"crossref","unstructured":"Shen T, Zhou T, Long G, Jiang J, Pan S, Zhang C (2018) Disan: directional self-attention network for rnn\/cnn-free language understanding[C]. Proceedings of the AAAI Conference on Artificial Intelligence 32(1)","DOI":"10.1609\/aaai.v32i1.11941"},{"key":"1730_CR95","doi-asserted-by":"crossref","unstructured":"Al-Rfou R, Choe D, Constant N, Guo M, Jones L (2019) Character-level language modeling with deeper self-attention[C]. Proceedings of the AAAI Conference on Artificial Intelligence 33:3159\u20133166","DOI":"10.1609\/aaai.v33i01.33013159"},{"key":"1730_CR96","unstructured":"Huang C Z A, Vaswani A, Uszkoreit J (2018) An improved relative self-attention mechanism for transformer with application to music generation [J].\u00a0CoRR\u00a0abs\/1809.04281"},{"key":"1730_CR97","doi-asserted-by":"crossref","unstructured":"Shen T, Zhou T, Long G, Jiang J, Wang S, Zhang C (2018) Reinforced self-attention network: a hybrid of hard and soft attention for sequence modeling[J].\u00a0arXiv preprint arXiv:1801.10296","DOI":"10.24963\/ijcai.2018\/604"},{"key":"1730_CR98","doi-asserted-by":"crossref","unstructured":"Tang G, M\u00fcller M, Rios A, Sennrich R \u00a0(2018) Why self-attention? A targeted evaluation of neural machine translation architectures[J]. arXiv preprint arXiv:1808.08946","DOI":"10.18653\/v1\/D18-1458"},{"key":"1730_CR99","doi-asserted-by":"crossref","unstructured":"Du J, Han J, Way A, Wan D (2018) Multi-level structured self-attentions for distantly supervised relation extraction[C]. Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing, 2216\u20132225","DOI":"10.18653\/v1\/D18-1245"},{"key":"1730_CR100","unstructured":"Weston J, Chopra S, Bordes A (2015) Memory networks[C]. International Conference on Learning Representations"},{"key":"1730_CR101","unstructured":"Sukhbaatar S, Weston J, Fergus R\u00a0(2015)\u00a0End-to-end memory networks[C]. Advances in Neural Information Processing Systems 2440\u20132448"},{"key":"1730_CR102","unstructured":"Kumar A, Irsoy O, Ondruska P\u00a0(2016)\u00a0Ask me anything: dynamic memory networks for natural language processing[C]. International Conference on Machine Learning 1378\u20131387"},{"key":"1730_CR103","unstructured":"Goodfellow I, Pouget-Abadie J, Mirza M (2014)\u00a0Generative adversarial nets[C]. Advances in Neural Information Processing Systems 2672\u20132680"},{"key":"1730_CR104","doi-asserted-by":"crossref","unstructured":"Rohrbach A, Rohrbach M, Tandon N, Schiele B\u00a0(2015)\u00a0A dataset for movie description[C]. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition 3202\u20133212","DOI":"10.1109\/CVPR.2015.7298940"},{"key":"1730_CR105","doi-asserted-by":"publisher","first-page":"9127","DOI":"10.1609\/aaai.v33i01.33019127","volume":"33","author":"Z Yu","year":"2019","unstructured":"Yu Z, Xu D, Yu J, Yu T, Zhao Z, Zhuang Y, Tao D (2019) ActivityNet-QA: a dataset for understanding complex web videos via question answering[C]. Proceedings of the AAAI Conference on Artificial Intelligence 33:9127\u20139134","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"1730_CR106","doi-asserted-by":"crossref","unstructured":"Li Y, Song Y, Cao L\u00a0(2016)\u00a0Tetreault J, Goldberg L, Jaimes A, Luo J TGIF: a new dataset and benchmark on animated GIF description[C]. Proc IEEE Conf Comput Vis Pattern Recognit 4641\u20134650","DOI":"10.1109\/CVPR.2016.502"},{"key":"1730_CR107","doi-asserted-by":"crossref","unstructured":"Wu Z, Palmer M\u00a0(1994)\u00a0Verbs semantics and lexical selection[C]. In Proceedings of the 32nd annual meeting on Association for Computational Linguistics 133\u2013138","DOI":"10.3115\/981732.981751"},{"key":"1730_CR108","doi-asserted-by":"crossref","unstructured":"Fellbaum C\u00a0(1998)\u00a0Towards a representation of idioms in WordNet[C]. Processing Systems","DOI":"10.7551\/mitpress\/7287.001.0001"},{"key":"1730_CR109","doi-asserted-by":"crossref","unstructured":"Liu C N , Chen D J , Chen H T , Liu T L (2018) A2A: attention to attention reasoning for movie question answering[C]. Asian Conference on Computer Vision 404\u2013419","DOI":"10.1007\/978-3-030-20876-9_26"}],"container-title":["Mobile Networks and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11036-020-01730-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11036-020-01730-0\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11036-020-01730-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,12,13]],"date-time":"2022-12-13T07:54:51Z","timestamp":1670918091000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11036-020-01730-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,1,25]]},"references-count":109,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2021,10]]}},"alternative-id":["1730"],"URL":"https:\/\/doi.org\/10.1007\/s11036-020-01730-0","relation":{},"ISSN":["1383-469X","1572-8153"],"issn-type":[{"value":"1383-469X","type":"print"},{"value":"1572-8153","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021,1,25]]},"assertion":[{"value":"9 December 2020","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 January 2021","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}