{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,21]],"date-time":"2025-10-21T15:43:35Z","timestamp":1761061415636,"version":"3.37.3"},"reference-count":34,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2021,6,2]],"date-time":"2021-06-02T00:00:00Z","timestamp":1622592000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,6,2]],"date-time":"2021-06-02T00:00:00Z","timestamp":1622592000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61876130","61932009"],"award-info":[{"award-number":["61876130","61932009"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2022,2]]},"DOI":"10.1007\/s00530-021-00805-6","type":"journal-article","created":{"date-parts":[[2021,6,2]],"date-time":"2021-06-02T18:48:58Z","timestamp":1622659738000},"page":"161-169","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":8,"title":["Complementary spatiotemporal network for video question answering"],"prefix":"10.1007","volume":"28","author":[{"given":"Xinrui","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Aming","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2768-1398","authenticated-orcid":false,"given":"Yahong","family":"Han","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2021,6,2]]},"reference":[{"key":"805_CR1","unstructured":"Ba, J.L., Kiros, J.R., Hinton, G.E.: Layer normalization. arXiv:1607.06450 (2016)"},{"key":"805_CR2","unstructured":"Chung, J., Gulcehre, C., Cho, K., Bengio, Y.: Empirical evaluation of gated recurrent neural networks on sequence modeling. arXiv:1412.3555 (2014)"},{"key":"805_CR3","doi-asserted-by":"crossref","unstructured":"Dai, Z., Yang, Z., Yang, Y., Carbonell, J., Le, Q.V., Salakhutdinov, R.: Transformer-xl: attentive language models beyond a fixed-length context. arXiv:1901.02860 (2019)","DOI":"10.18653\/v1\/P19-1285"},{"key":"805_CR4","doi-asserted-by":"crossref","unstructured":"Donahue, J., Anne Hendricks, L., Guadarrama, S., Rohrbach, M., Venugopalan, S., Saenko, K., Darrell, T.: Long-term recurrent convolutional networks for visual recognition and description. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 2625\u20132634 (2015)","DOI":"10.1109\/CVPR.2015.7298878"},{"key":"805_CR5","doi-asserted-by":"crossref","unstructured":"Fan, C., Zhang, X., Zhang, S., Wang, W., Zhang, C., Huang, H.: Heterogeneous memory enhanced multimodal attention model for video question answering. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 1999\u20132007 (2019)","DOI":"10.1109\/CVPR.2019.00210"},{"key":"805_CR6","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., Fan, H., Malik, J., He, K.: Slowfast networks for video recognition. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp. 6202\u20136211 (2019)","DOI":"10.1109\/ICCV.2019.00630"},{"key":"805_CR7","doi-asserted-by":"crossref","unstructured":"Gao, J., Ge, R., Chen, K., Nevatia, R.: Motion-appearance co-memory networks for video question answering. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 6576\u20136585 (2018)","DOI":"10.1109\/CVPR.2018.00688"},{"key":"805_CR8","unstructured":"Ghiasi, G., Lin, T.Y., Le, Q.V.: Dropblock: A regularization method for convolutional networks. arXiv:1810.12890 (2018)"},{"key":"805_CR9","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Doll\u00e1r, P., Girshick, R.: Mask r-cnn. In: Proceedings of the IEEE international conference on computer vision, pp. 2961\u20132969 (2017)","DOI":"10.1109\/ICCV.2017.322"},{"key":"805_CR10","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"805_CR11","doi-asserted-by":"crossref","unstructured":"Hou, R., Chang, H., Ma, B., Shan, S., Chen, X.: Temporal complementary learning for video person re-identification. In: European conference on computer vision, pp. 388\u2013405. Springer (2020)","DOI":"10.1007\/978-3-030-58595-2_24"},{"key":"805_CR12","first-page":"11021","volume":"34","author":"D Huang","year":"2020","unstructured":"Huang, D., Chen, P., Zeng, R., Du, Q., Tan, M., Gan, C.: Location-aware graph convolutional networks for video question answering. Proc. AAAI Conf. Artif. Intell. 34, 11021\u201311028 (2020)","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"key":"805_CR13","doi-asserted-by":"crossref","unstructured":"Jang, Y., Song, Y., Yu, Y., Kim, Y., Kim, G.: Tgif-qa: toward spatio-temporal reasoning in visual question answering. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 2758\u20132766 (2017)","DOI":"10.1109\/CVPR.2017.149"},{"key":"805_CR14","doi-asserted-by":"crossref","unstructured":"Jin, W., Zhao, Z., Gu, M., Yu, J., Xiao, J., Zhuang, Y.: Multi-interaction network with object relation for video question answering. In: Proceedings of the 27th ACM international conference on multimedia, pp. 1193\u20131201 (2019)","DOI":"10.1145\/3343031.3351065"},{"key":"805_CR15","doi-asserted-by":"crossref","unstructured":"Kim, K.M., Heo, M.O., Choi, S.H., Zhang, B.T.: Deepstory: video story qa by deep embedded memory networks. arXiv:1707.00836 (2017)","DOI":"10.24963\/ijcai.2017\/280"},{"key":"805_CR16","unstructured":"Kipf, T.N., Welling, M.: Semi-supervised classification with graph convolutional networks. arXiv:1609.02907 (2016)"},{"key":"805_CR17","doi-asserted-by":"crossref","unstructured":"Le, T.M., Le, V., Venkatesh, S., Tran, T.: Hierarchical conditional relation networks for video question answering. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 9972\u20139981 (2020)","DOI":"10.1109\/CVPR42600.2020.00999"},{"key":"805_CR18","doi-asserted-by":"crossref","unstructured":"Lei, J., Wang, L., Shen, Y., Yu, D., Berg, T.L., Bansal, M.: Mart: memory-augmented recurrent transformer for coherent video paragraph captioning. arXiv:2005.05402 (2020)","DOI":"10.18653\/v1\/2020.acl-main.233"},{"key":"805_CR19","doi-asserted-by":"crossref","unstructured":"Lei, J., Yu, L., Bansal, M., Berg, T.L.: Tvqa: Localized, compositional video question answering. arXiv:1809.01696 (2018)","DOI":"10.18653\/v1\/D18-1167"},{"key":"805_CR20","doi-asserted-by":"crossref","unstructured":"Lei, J., Yu, L., Berg, T.L., Bansal, M.: Tvr: a large-scale dataset for video-subtitle moment retrieval. arXiv:2001.09099 (2020)","DOI":"10.1007\/978-3-030-58589-1_27"},{"key":"805_CR21","first-page":"8658","volume":"33","author":"X Li","year":"2019","unstructured":"Li, X., Song, J., Gao, L., Liu, X., Huang, W., He, X., Gan, C.: Beyond rnns: positional self-attention with co-attention for video question answering. Proc. AAAI Conf. Artif. Intell. 33, 8658\u20138665 (2019)","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"key":"805_CR22","doi-asserted-by":"crossref","unstructured":"Shaw, P., Uszkoreit, J., Vaswani, A.: Self-attention with relative position representations. arXiv:1803.02155 (2018)","DOI":"10.18653\/v1\/N18-2074"},{"issue":"1","key":"805_CR23","first-page":"1929","volume":"15","author":"N Srivastava","year":"2014","unstructured":"Srivastava, N., Hinton, G., Krizhevsky, A., Sutskever, I., Salakhutdinov, R.: Dropout: a simple way to prevent neural networks from overfitting. J. Mach. Learn. Res. 15(1), 1929\u20131958 (2014)","journal-title":"J. Mach. Learn. Res."},{"key":"805_CR24","unstructured":"Sukhbaatar, S., Szlam, A., Weston, J., Fergus, R.: End-to-end memory networks. arXiv:1503.08895 (2015)"},{"key":"805_CR25","doi-asserted-by":"crossref","unstructured":"Tran, D., Bourdev, L., Fergus, R., Torresani, L., Paluri, M.: Learning spatiotemporal features with 3d convolutional networks. In: Proceedings of the IEEE international conference on computer vision, pp. 4489\u20134497 (2015)","DOI":"10.1109\/ICCV.2015.510"},{"key":"805_CR26","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, L., Polosukhin, I.: Attention is all you need. arXiv:1706.03762 (2017)"},{"key":"805_CR27","doi-asserted-by":"crossref","unstructured":"Xu, D., Zhao, Z., Xiao, J., Wu, F., Zhang, H., He, X., Zhuang, Y.: Video question answering via gradually refined attention over appearance and motion. In: Proceedings of the 25th ACM international conference on Multimedia, pp. 1645\u20131653 (2017)","DOI":"10.1145\/3123266.3123427"},{"issue":"4","key":"805_CR28","doi-asserted-by":"publisher","first-page":"1445","DOI":"10.1109\/TPAMI.2020.2975798","volume":"43","author":"C Yan","year":"2021","unstructured":"Yan, C., Gong, B., Wei, Y., Gao, Y.: Deep multi-view enhancement hashing for image retrieval. IEEE Trans. Pattern Anal. Mach. Intel. 43(4), 1445\u20131451 (2021)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intel."},{"key":"805_CR29","unstructured":"Yan, C., Hao, Y., Li, L., Yin, J., Liu, A., Mao, Z., Chen, Z., Gao, X.: Task-adaptive attention for image captioning. IEEE Trans. Circ. Syst. Video Technol. 14(8) (2021)"},{"issue":"4","key":"805_CR30","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3404374","volume":"16","author":"C Yan","year":"2020","unstructured":"Yan, C., Li, Z., Zhang, Y., Liu, Y., Ji, X., Zhang, Y.: Depth image denoising using nuclear norm and learning graph model. ACM Trans. Multimedia Comput. Commun. Appl. 16(4), 1\u201317 (2020)","journal-title":"ACM Trans. Multimedia Comput. Commun. Appl."},{"issue":"11","key":"805_CR31","doi-asserted-by":"publisher","first-page":"3014","DOI":"10.1109\/TMM.2020.2967645","volume":"22","author":"C Yan","year":"2020","unstructured":"Yan, C., Shao, B., Zhao, H., Ning, R., Zhang, Y., Xu, F.: 3d room layout estimation from a single rgb image. IEEE Trans. Multimedia 22(11), 3014\u20133024 (2020)","journal-title":"IEEE Trans. Multimedia"},{"issue":"12","key":"805_CR32","doi-asserted-by":"publisher","first-page":"5939","DOI":"10.1109\/TIP.2019.2922062","volume":"28","author":"Z Zhao","year":"2019","unstructured":"Zhao, Z., Zhang, Z., Xiao, S., Xiao, Z., Yan, X., Yu, J., Cai, D., Wu, F.: Long-form video question answering via dynamic hierarchical reinforced networks. IEEE Trans. Image Process. 28(12), 5939\u20135952 (2019)","journal-title":"IEEE Trans. Image Process."},{"key":"805_CR33","doi-asserted-by":"crossref","unstructured":"Zhou, L., Zhou, Y., Corso, J.J., Socher, R., Xiong, C.: End-to-end dense video captioning with masked transformer. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 8739\u20138748 (2018)","DOI":"10.1109\/CVPR.2018.00911"},{"key":"805_CR34","doi-asserted-by":"crossref","unstructured":"Zhu, L., Yang, Y.: Actbert: learning global-local video-text representations. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 8746\u20138755 (2020)","DOI":"10.1109\/CVPR42600.2020.00877"}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-021-00805-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-021-00805-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-021-00805-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,1,28]],"date-time":"2022-01-28T06:10:48Z","timestamp":1643350248000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-021-00805-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,6,2]]},"references-count":34,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2022,2]]}},"alternative-id":["805"],"URL":"https:\/\/doi.org\/10.1007\/s00530-021-00805-6","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"type":"print","value":"0942-4962"},{"type":"electronic","value":"1432-1882"}],"subject":[],"published":{"date-parts":[[2021,6,2]]},"assertion":[{"value":"29 March 2021","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"30 April 2021","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 June 2021","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}