{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,19]],"date-time":"2026-01-19T07:04:33Z","timestamp":1768806273267,"version":"3.49.0"},"reference-count":62,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2021,2,12]],"date-time":"2021-02-12T00:00:00Z","timestamp":1613088000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,2,12]],"date-time":"2021-02-12T00:00:00Z","timestamp":1613088000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2021,5]]},"DOI":"10.1007\/s11263-020-01409-9","type":"journal-article","created":{"date-parts":[[2021,2,12]],"date-time":"2021-02-12T18:22:34Z","timestamp":1613154154000},"page":"1484-1505","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":30,"title":["Spatial\u2013Temporal Relation Reasoning for Action Prediction in Videos"],"prefix":"10.1007","volume":"129","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2056-6947","authenticated-orcid":false,"given":"Xinxiao","family":"Wu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ruiqi","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jingyi","family":"Hou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hanxi","family":"Lin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiebo","family":"Luo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2021,2,12]]},"reference":[{"key":"1409_CR1","doi-asserted-by":"crossref","unstructured":"Aditya, S., Yang, Y., & Baral, C. (2018). Explicit reasoning over end-to-end neural architectures for visual question answering. In Thirty-second AAAI conference on artificial intelligence.","DOI":"10.1609\/aaai.v32i1.11324"},{"key":"1409_CR2","doi-asserted-by":"crossref","unstructured":"Aliakbarian, M. S., Saleh, F. S., Salzmann, M., Fernando, B., Petersson, L., & Andersson, L. (2017). Encouraging lstms to anticipate actions very early. In Proceedings of the IEEE conference on computer vision and pattern recognition.","DOI":"10.1109\/ICCV.2017.39"},{"key":"1409_CR3","doi-asserted-by":"crossref","unstructured":"Anderson, P., He, X., Buehler, C., Teney, D., Johnson, M., Gould, S., et al. (2018). Bottom-up and top-down attention for image captioning and visual question answering. In Proceedings of the IEEE conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR.2018.00636"},{"key":"1409_CR4","unstructured":"Arthur, D., & Vassilvitskii, S. (2007). K-means++: The advantages of careful seeding. In Eighteenth ACM-SIAM symposium on discrete algorithms."},{"key":"1409_CR5","unstructured":"Bhoi, A. (2019). Spatio-temporal action recognition: A survey. arXiv preprint arXiv:1901.09403."},{"key":"1409_CR6","doi-asserted-by":"crossref","unstructured":"Cai, Y., Li, H., Hu, J. F., & Zheng, W. S. (2019). Action knowledge transfer for action prediction with partial videos. In Proceedings of the AAAI conference on artificial intelligence.","DOI":"10.1609\/aaai.v33i01.33018118"},{"key":"1409_CR7","doi-asserted-by":"crossref","unstructured":"Cao, Y., Barrett, D., Barbu, A., Narayanaswamy, S., Yu, H., Michaux, A., et al. (2013). Recognize human activities from partially observed videos. In Proceedings of the IEEE conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR.2013.343"},{"key":"1409_CR8","doi-asserted-by":"crossref","unstructured":"Chen, L., Lu, J., Song, Z., & Zhou, J. (2018a). Part-activated deep reinforcement learning for action prediction. In European conference on computer vision.","DOI":"10.1007\/978-3-030-01219-9_26"},{"key":"1409_CR9","doi-asserted-by":"crossref","unstructured":"Chen, X., Li, L. J., Fei-Fei, L., & Gupta, A. (2018b). Iterative visual reasoning beyond convolutions. In Proceedings of the IEEE conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR.2018.00756"},{"key":"1409_CR10","doi-asserted-by":"crossref","unstructured":"Cho, K., van Merrienboer, B., G\u00fcl\u00e7ehre, \u00c7., Bahdanau, D., Bougares, F., Schwenk, H., et al. (2014). Learning phrase representations using RNN encoder-decoder for statistical machine translation. In Proceedings of the 2014 conference on empirical methods in natural language processing.","DOI":"10.3115\/v1\/D14-1179"},{"key":"1409_CR11","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L. J., Li, K., & Fei-Fei, L. (2009). Imagenet: A large-scale hierarchical image database. In Proceedings of the IEEE conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR.2009.5206848"},{"issue":"1\u20132","key":"1409_CR12","doi-asserted-by":"publisher","first-page":"165","DOI":"10.1016\/0010-0277(93)90039-X","volume":"49","author":"JSB Evans","year":"1993","unstructured":"Evans, J. S. B., Over, D. E., & Manktelow, K. I. (1993). Reasoning, decision making and rationality. Cognition, 49(1\u20132), 165\u2013187.","journal-title":"Cognition"},{"key":"1409_CR13","unstructured":"Gilmer, J., Schoenholz, S. S., Riley, P. F., Vinyals, O., & Dahl, G. E. (2017). Neural message passing for quantum chemistry. In Proceedings of the 34th international conference on machine learning."},{"key":"1409_CR14","doi-asserted-by":"crossref","unstructured":"Girshick, R. (2015). Fast r-CNN. In Proceedings of the IEEE international conference on computer vision.","DOI":"10.1109\/ICCV.2015.169"},{"key":"1409_CR15","unstructured":"Girshick, R., Radosavovic, I., Gkioxari, G., Doll\u00e1r, P., & He, K. (2018). Detectron. https:\/\/github.com\/facebookresearch\/detectron."},{"key":"1409_CR16","doi-asserted-by":"crossref","unstructured":"Goyal, R., Kahou, S. E., Michalski, V., Materzynska, J., Westphal, S., Kim, H., et\u00a0al. (2017). The \u201csomething something\u201d video database for learning and evaluating visual common sense. In Proceedings of the IEEE international conference on computer vision.","DOI":"10.1109\/ICCV.2017.622"},{"key":"1409_CR17","unstructured":"Hanwang, Z., Kyaw, Z., Chang, S., & Chua, T. (2017). Visual translation embedding network for visual relation detection. In Proceedings of the IEEE conference on computer vision and pattern recognition."},{"key":"1409_CR18","doi-asserted-by":"crossref","unstructured":"Hara, K., Kataoka, H., & Satoh, Y. (2018). Can spatiotemporal 3d CNNs retrace the history of 2d CNNs and imagenet? In Proceedings of the IEEE conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR.2018.00685"},{"key":"1409_CR19","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., & Sun, J. (2016). Deep residual learning for image recognition. In Proceedings of the IEEE conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR.2016.90"},{"key":"1409_CR20","doi-asserted-by":"crossref","unstructured":"Herzig, R., Levi, E., Xu, H., Gao, H., Brosh, E., Wang, X., et al. (2019). Spatio-temporal action graph networks. In Proceedings of the IEEE international conference on computer vision workshops.","DOI":"10.1109\/ICCVW.2019.00288"},{"issue":"11","key":"1409_CR21","doi-asserted-by":"publisher","first-page":"2568","DOI":"10.1109\/TPAMI.2018.2863279","volume":"41","author":"JF Hu","year":"2018","unstructured":"Hu, J. F., Zheng, W. S., Ma, L., Wang, G., Lai, J., & Zhang, J. (2018). Early action prediction by soft regression. IEEE Transactions on Pattern Analysis and Machine Intelligence, 41(11), 2568\u20132583.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"1409_CR22","unstructured":"Ioffe, S., & Szegedy, C. (2015). Batch normalization: Accelerating deep network training by reducing internal covariate shift. In International conference on machine learning."},{"issue":"1","key":"1409_CR23","doi-asserted-by":"publisher","first-page":"221","DOI":"10.1109\/TPAMI.2012.59","volume":"35","author":"S Ji","year":"2013","unstructured":"Ji, S., Xu, W., Yang, M., & Yu, K. (2013). 3d convolutional neural networks for human action recognition. IEEE Transactions on Pattern Analysis and Machine Intelligence, 35(1), 221\u2013231.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"1409_CR24","unstructured":"Kay, W., Carreira, J., Simonyan, K., Zhang, B., Hillier, C., Vijayanarasimhan, S., et\u00a0al. (2017). The kinetics human action video dataset. arXiv preprint arXiv:1705.06950."},{"key":"1409_CR25","unstructured":"Kingma, D. P., & Ba, J. (2015). Adam: A method for stochastic optimization. In 3rd International conference on learning representations."},{"key":"1409_CR26","unstructured":"Kipf, T. N., & Welling, M. (2017). Semi-supervised classification with graph convolutional networks. In 5th International conference on learning representations."},{"key":"1409_CR27","doi-asserted-by":"publisher","first-page":"1844","DOI":"10.1109\/TPAMI.2015.2491928","volume":"9","author":"Y Kong","year":"2016","unstructured":"Kong, Y., & Fu, Y. (2016). Max-margin action prediction machine. IEEE Transactions on Pattern Analysis and Machine Intelligence, 9, 1844\u20131858.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"1409_CR28","doi-asserted-by":"crossref","unstructured":"Kong, Y., Gao, S., Sun, B., & Fu, Y. (2018). Action prediction from videos via memorizing hard-to-predict samples. In AAAI conference on artificial intelligence.","DOI":"10.1609\/aaai.v32i1.12324"},{"key":"1409_CR29","doi-asserted-by":"publisher","first-page":"1775","DOI":"10.1109\/TPAMI.2014.2303090","volume":"9","author":"Y Kong","year":"2014","unstructured":"Kong, Y., Jia, Y., & Fu, Y. (2014a). Interactive phrases: Semantic descriptionsfor human interaction recognition. IEEE Transactions on Pattern Analysis and Machine Intelligence, 9, 1775\u20131788.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"1409_CR30","doi-asserted-by":"crossref","unstructured":"Kong, Y., Kit, D., & Fu, Y. (2014b). A discriminative model with multiple temporal scales for action prediction. In European conference on computer vision.","DOI":"10.1007\/978-3-319-10602-1_39"},{"key":"1409_CR31","doi-asserted-by":"crossref","unstructured":"Kong, Y., Tao, Z., & Fu, Y. (2017). Deep sequential context networks for action prediction. In Proceedings of the IEEE conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR.2017.390"},{"issue":"3","key":"1409_CR32","doi-asserted-by":"publisher","first-page":"539","DOI":"10.1109\/TPAMI.2018.2882805","volume":"42","author":"Y Kong","year":"2020","unstructured":"Kong, Y., Tao, Z., & Fu, Y. (2020). Adversarial action prediction networks. IEEE Transactions on Pattern Analysis and Machine Intelligence, 42(3), 539\u2013553.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"issue":"8","key":"1409_CR33","doi-asserted-by":"publisher","first-page":"951","DOI":"10.1177\/0278364913478446","volume":"32","author":"HS Koppula","year":"2013","unstructured":"Koppula, H. S., Gupta, R., & Saxena, A. (2013). Learning human activities and object affordances from rgb-d videos. The International Journal of Robotics Research, 32(8), 951\u2013970.","journal-title":"The International Journal of Robotics Research"},{"key":"1409_CR34","doi-asserted-by":"crossref","unstructured":"Kuehne, H., Jhuang, H., Garrote, E., Poggio, T., & Serre, T. (2011). HMDB: A large video database for human motion recognition. In Proceedings of the international conference on computer vision.","DOI":"10.1109\/ICCV.2011.6126543"},{"issue":"5","key":"1409_CR35","doi-asserted-by":"publisher","first-page":"2272","DOI":"10.1109\/TIP.2017.2751145","volume":"27","author":"S Lai","year":"2018","unstructured":"Lai, S., Zheng, W. S., Hu, J. F., & Zhang, J. (2018). Global-local temporal saliency action prediction. IEEE Transactions on Image Process, 27(5), 2272\u20132285.","journal-title":"IEEE Transactions on Image Process"},{"key":"1409_CR36","doi-asserted-by":"crossref","unstructured":"Lan, T., Chen, T. C., & Savarese, S. (2014). A hierarchical representation for future action prediction. In European conference on computer vision (pp. 689\u2013704).","DOI":"10.1007\/978-3-319-10578-9_45"},{"issue":"8","key":"1409_CR37","doi-asserted-by":"publisher","first-page":"1644","DOI":"10.1109\/TPAMI.2013.2297321","volume":"36","author":"K Li","year":"2014","unstructured":"Li, K., & Fu, Y. (2014). Prediction of human activity by discovering temporal sequence patterns. IEEE Transactions on Pattern Analysis and Machine Intelligence, 36(8), 1644\u20131657.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"1409_CR38","unstructured":"Li, Y., Tarlow, D., Brockschmidt, M., & Zemel, R. S. (2016). Gated graph sequence neural networks. In 4th International conference on learning representations."},{"key":"1409_CR39","doi-asserted-by":"crossref","unstructured":"Liang, K., Guo, Y., Chang, H., & Chen, X. (2018). Visual relationship detection with deep structural ranking. In AAAI conference on artificial intelligence.","DOI":"10.1609\/aaai.v32i1.12274"},{"key":"1409_CR40","doi-asserted-by":"crossref","unstructured":"Liao, W., Rosenhahn, B., Shuai, L., & Ying\u00a0Yang, M. (2019). Natural language guided visual relationship detection. In Proceedings of the IEEE conference on computer vision and pattern recognition workshops.","DOI":"10.1109\/CVPRW.2019.00058"},{"key":"1409_CR41","doi-asserted-by":"crossref","unstructured":"Liu, J., Luo, J., & Shah, M. (2009). Recognizing realistic actions from videos. In Proceedings of the IEEE conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR.2009.5206744"},{"key":"1409_CR42","doi-asserted-by":"crossref","unstructured":"Lu, C., Krishna, R., Bernstein, M., & Fei-Fei, L. (2016). Visual relationship detection with language priors. In European conference on computer vision.","DOI":"10.1007\/978-3-319-46448-0_51"},{"key":"1409_CR43","unstructured":"Newell, A., & Deng, J. (2017). Pixels to graphs by associative embedding. In Advances in neural information processing systems."},{"key":"1409_CR44","unstructured":"Nicolicioiu, A., Duta, I., & Leordeanu, M. (2019). Recurrent space-time graph neural networks. In Advances in neural information processing systems."},{"key":"1409_CR45","doi-asserted-by":"crossref","unstructured":"Pang, G., Wang, X., Hu, J. F., Zhang, Q., & Zheng, W. S. (2019). Dbdnet: Learning bi-directional dynamics for early action prediction. In Proceedings of the 28th international joint conference on artificial intelligence.","DOI":"10.24963\/ijcai.2019\/126"},{"key":"1409_CR46","doi-asserted-by":"crossref","unstructured":"Qi, M., Li, W., Yang, Z., Wang, Y., & Luo, J. (2019). Attentive relational networks for mapping images to scene graphs. In Proceedings of the IEEE conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR.2019.00408"},{"key":"1409_CR47","doi-asserted-by":"crossref","unstructured":"Ryoo, M. S. (2011). Human activity prediction: Early recognition of ongoing activities from streaming videos. In IEEE international conference on computer vision.","DOI":"10.1109\/ICCV.2011.6126349"},{"issue":"1","key":"1409_CR48","doi-asserted-by":"publisher","first-page":"61","DOI":"10.1109\/TNN.2008.2005605","volume":"20","author":"F Scarselli","year":"2008","unstructured":"Scarselli, F., Gori, M., Tsoi, A. C., Hagenbuchner, M., & Monfardini, G. (2008). The graph neural network model. IEEE Transactions on Neural Networks, 20(1), 61\u201380.","journal-title":"IEEE Transactions on Neural Networks"},{"key":"1409_CR49","doi-asserted-by":"crossref","unstructured":"Shang, X., Ren, T., Guo, J., Zhang, H., & Tat-Seng, C. (2017). Video visual relation detection. In Proceedings of the 25th ACM international conference on multimedia.","DOI":"10.1145\/3123266.3123380"},{"key":"1409_CR50","doi-asserted-by":"crossref","unstructured":"Si, C., Jing, Y., Wang, W., Wang, L., & Tan, T. (2018). Skeleton-based action recognition with spatial reasoning and temporal stack learning. In Proceedings of the European conference on computer vision.","DOI":"10.1007\/978-3-030-01246-5_7"},{"key":"1409_CR51","unstructured":"Soomro, K., Zamir, A. R., & Shah, M. (2012). Ucf101: A dataset of 101 human actions classes from videos in the wild. arXiv preprint arXiv:1212.0402."},{"key":"1409_CR52","doi-asserted-by":"crossref","unstructured":"Sun, C., Shrivastava, A., Vondrick, C., Sukthankar, R., Murphy, K., & Schmid, C. (2019). Relational action forecasting. In Proceedings of the IEEE conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR.2019.00036"},{"key":"1409_CR53","doi-asserted-by":"crossref","unstructured":"Tsai, Y. H. H., Divvala, S., Morency, L. P., Salakhutdinov, R., & Farhadi, A. (2019). Video relationship reasoning using gated spatio-temporal energy graph. In Proceedings of the IEEE conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR.2019.01067"},{"key":"1409_CR54","unstructured":"Veli\u010dkovi\u0107, P., Cucurull, G., Casanova, A., Romero, A., Lio, P., & Bengio, Y. (2018). Graph attention networks. In ICLR."},{"key":"1409_CR55","doi-asserted-by":"crossref","unstructured":"Wang, L., Xiong, Y., Wang, Z., Qiao, Y., Lin, D., Tang, X., et al. (2016). Temporal segment networks: Towards good practices for deep action recognition. In European conference on computer vision.","DOI":"10.1007\/978-3-319-46484-8_2"},{"key":"1409_CR56","doi-asserted-by":"crossref","unstructured":"Wang, X., & Gupta, A. (2018). Videos as space-time region graphs. In Proceedings of the European conference on computer vision.","DOI":"10.1007\/978-3-030-01228-1_25"},{"key":"1409_CR57","doi-asserted-by":"crossref","unstructured":"Wang, X., Hu, J. F., Lai, J. H., Zhang, J., & Zheng, W. S. (2019). Progressive teacher-student learning for early action prediction. In Proceedings of the IEEE conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR.2019.00367"},{"key":"1409_CR58","unstructured":"Woo, S., Kim, D., Cho, D., & Kweon, I. S. (2018). Linknet: Relational embedding for scene graph. In Advances in neural information processing systems."},{"key":"1409_CR59","doi-asserted-by":"crossref","unstructured":"Xu, H., Jiang, C., Liang, X., & Li, Z. (2019). Spatial-aware graph relation network for large-scale object detection. In Proceedings of the IEEE conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR.2019.00952"},{"key":"1409_CR60","doi-asserted-by":"crossref","unstructured":"Zhang, J., Kalantidis, Y., Rohrbach, M., Paluri, M., Elgammal, A., & Elhoseiny, M. (2019). Large-scale visual relationship understanding. In Proceedings of the AAAI conference on artificial intelligence.","DOI":"10.1609\/aaai.v33i01.33019185"},{"key":"1409_CR61","doi-asserted-by":"crossref","unstructured":"Zhao, H., & Wildes, R. P. (2019). Spatiotemporal feature residual propagation for action prediction. In Proceedings of the IEEE international conference on computer vision.","DOI":"10.1109\/ICCV.2019.00710"},{"key":"1409_CR62","doi-asserted-by":"crossref","unstructured":"Zhou, B., Andonian, A., Oliva, A., & Torralba, A. (2018). Temporal relational reasoning in videos. In Proceedings of the European conference on computer vision.","DOI":"10.1007\/978-3-030-01246-5_49"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-020-01409-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-020-01409-9\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-020-01409-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,12,17]],"date-time":"2022-12-17T02:02:18Z","timestamp":1671242538000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-020-01409-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,2,12]]},"references-count":62,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2021,5]]}},"alternative-id":["1409"],"URL":"https:\/\/doi.org\/10.1007\/s11263-020-01409-9","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021,2,12]]},"assertion":[{"value":"11 December 2019","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 November 2020","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 February 2021","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}