{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,20]],"date-time":"2025-11-20T12:45:49Z","timestamp":1763642749196},"reference-count":39,"publisher":"Institute of Electronics, Information and Communications Engineers (IEICE)","issue":"12","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEICE Trans. Inf. &amp; Syst."],"published-print":{"date-parts":[[2019,12,1]]},"DOI":"10.1587\/transinf.2019edp7104","type":"journal-article","created":{"date-parts":[[2019,12,2]],"date-time":"2019-12-02T16:27:54Z","timestamp":1575304074000},"page":"2568-2576","source":"Crossref","is-referenced-by-count":5,"title":["Attentive Sequences Recurrent Network for Social Relation Recognition from Video"],"prefix":"10.1587","volume":"E102.D","author":[{"given":"Jinna","family":"LV","sequence":"first","affiliation":[{"name":"Beijing University of Posts and Telecommunications"},{"name":"Beijing Information Science & Technology University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bin","family":"WU","sequence":"additional","affiliation":[{"name":"Beijing University of Posts and Telecommunications"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yunlei","family":"ZHANG","sequence":"additional","affiliation":[{"name":"Beijing University of Posts and Telecommunications"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yunpeng","family":"XIAO","sequence":"additional","affiliation":[{"name":"Chongqing University of Posts and Telecommunications"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"532","reference":[{"key":"1","doi-asserted-by":"publisher","unstructured":"[1] Y.-J. Lin and S.-K. Weng, \u201cTrajectory Estimation of the Players and Shuttlecock for the Broadcast Badminton Videos,\u201d IEICE Transactions, vol.E101-A, no.10, pp.1730-1734, 2018. 10.1587\/transfun.e101.a.1730","DOI":"10.1587\/transfun.E101.A.1730"},{"key":"2","doi-asserted-by":"crossref","unstructured":"[2] A. Alahi, K. Goel, V. Ramanathan, A. Robicquet, L. Fei-Fei, and S. Savarese, \u201cSocial lstm: Human trajectory prediction in crowded spaces,\u201d Proc. Int. IEEE Conf. Computer Vision and Pattern Recognition, pp.961-971, 2016. 10.1109\/cvpr.2016.110","DOI":"10.1109\/CVPR.2016.110"},{"key":"3","doi-asserted-by":"publisher","unstructured":"[3] P.-S. Kim, D.-G. Lee, and S.-W. Lee, \u201cDiscriminative context learning with gated recurrent unit for group activity recognition,\u201d Pattern Recognition, vol.76, pp.149-161, 2018. 10.1016\/j.patcog.2017.10.037","DOI":"10.1016\/j.patcog.2017.10.037"},{"key":"4","doi-asserted-by":"publisher","unstructured":"[4] T. Lan, Y. Wang, W. Yang, S.N. Robinovitch, and G. Mori, \u201cDiscriminative latent models for recognizing contextual group activities,\u201d IEEE Trans. Pattern Anal. Mach. Intell, vol.34, no.8, pp.1549-1562, 2012. 10.1109\/tpami.2011.228","DOI":"10.1109\/TPAMI.2011.228"},{"key":"5","unstructured":"[5] Q.D. Tran and J.E. Jung, \u201cCocharnet: Extracting social networks using character co-occurrence in movies,\u201d Journal of Universal Computerence, vol.21, no.6, pp.796-815, 2015. 10.3217\/jucs-021-06-0796"},{"key":"6","doi-asserted-by":"publisher","unstructured":"[6] J. Lv, B. Wu, L. Zhou, and H. Wang, \u201cStoryRoleNet: Social Network Construction of Role Relationship in Video,\u201d IEEE Access, vol.6, pp.25958-25969, 2018. 10.1109\/access.2018.2832087","DOI":"10.1109\/ACCESS.2018.2832087"},{"key":"7","doi-asserted-by":"crossref","unstructured":"[7] Z. Zhang, P. Luo, C.-C. Loy, and X. Tang, \u201cLearning social relation traits from face images,\u201d Proc. IEEE Int. Conf. on Computer Vision, pp.3631-3639, 2015. 10.1109\/iccv.2015.414","DOI":"10.1109\/ICCV.2015.414"},{"key":"8","doi-asserted-by":"crossref","unstructured":"[8] Q. Sun, B. Schiele, and M. Fritz, \u201cA domain based approach to social relation recognition,\u201d Proc. IEEE Conf. Computer Vision and Pattern Recognition, pp.435-444, 2017. 10.1109\/cvpr.2017.54","DOI":"10.1109\/CVPR.2017.54"},{"key":"9","doi-asserted-by":"crossref","unstructured":"[9] M. Zhang, X. Liu, W. Liu, A. Zhou, H. Ma, and T. Mei, \u201cMulti-Granularity Reasoning for Social Relation Recognition from Images,\u201d CoRR ABZ\/1901.01067, 2019. 10.1109\/icme.2019.00279","DOI":"10.1109\/ICME.2019.00279"},{"key":"10","doi-asserted-by":"publisher","unstructured":"[10] C.-Y. Weng, W.-T. Chu, and J.-L. Wu, \u201cRoleNet: Movie analysis from the perspective of social networks,\u201d IEEE Trans. Multimedia, vol.11, no.2, pp.256-271, 2009. 10.1109\/tmm.2008.2009684","DOI":"10.1109\/TMM.2008.2009684"},{"key":"11","doi-asserted-by":"publisher","unstructured":"[11] S. Xia, M. Shao, J. Luo, and Y. Fu, \u201cUnderstanding kin relationships in a photo,\u201d IEEE Trans. Multimedia, vol.14, no.4, pp.1046-1056, 2012. 10.1109\/tmm.2012.2187436","DOI":"10.1109\/TMM.2012.2187436"},{"key":"12","doi-asserted-by":"publisher","unstructured":"[12] X. Tang, F. Guo, J. Shen, and T. Du, \u201cFacial Landmark Detection by Semi-supervised Deep Learning,\u201d Neurocomputing, vol.297, pp.22-32, 2018. 10.1016\/j.neucom.2018.01.080","DOI":"10.1016\/j.neucom.2018.01.080"},{"key":"13","doi-asserted-by":"publisher","unstructured":"[13] Z. Sun, Z.-P. Hu, R. Chiong, M. Wang, and W. He, \u201cCombining the Kernel Collaboration Representation and Deep Subspace Learning for Facial Expression Recognition,\u201d Journal of Circuits, Systems, and Computers, vol.27, no.8, pp.1-16, 2018. 10.1142\/s0218126618501219","DOI":"10.1142\/S0218126618501219"},{"key":"14","doi-asserted-by":"crossref","unstructured":"[14] J. Lv, W. Liu, L. Zhou, B. Wu, and H. Ma, \u201cMulti-stream Fusion Model for Social Relation Recognition from Videos,\u201d Proc. IEEE Conf. Conference on Multimedia Modeling, pp.355-368, 2018. 10.1007\/978-3-319-73603-7_29","DOI":"10.1007\/978-3-319-73603-7_29"},{"key":"15","unstructured":"[15] D. Tang, B. Qin, X. Feng, and T. Liu, \u201cEffective lstms for target-dependent sentiment classification,\u201d Coling, pp.3298-3307, 2016."},{"key":"16","doi-asserted-by":"publisher","unstructured":"[16] J. Song, Y. Guo, L. Gao, X. Li, A. Hanjalic, and H.T. Shen, \u201cFrom Deterministic to Generative: Multimodal Stochastic RNNs for Video Captioning,\u201d IEEE Transactions on Neural Networks and Learning Systems, vol.30, no.10, pp.3047-3058, 2019. 10.1109\/tnnls.2018.2851077","DOI":"10.1109\/TNNLS.2018.2851077"},{"key":"17","unstructured":"[17] Z. Huang, W. Xu, and K. Yu, \u201cBidirectional lstm-crf models for sequence tagging,\u201d CoRR abs\/1508.01991, pp.1-10, 2015."},{"key":"18","doi-asserted-by":"publisher","unstructured":"[18] L. Oksama and J. Hy\u00f6n\u00e4, \u201cDynamic binding of identity and location information: a serial model of multiple identity tracking,\u201d Cognitive Psychol, vol.56, no.4, pp.237-283, 2008. 10.1016\/j.cogpsych.2007.03.001","DOI":"10.1016\/j.cogpsych.2007.03.001"},{"key":"19","doi-asserted-by":"publisher","unstructured":"[19] A. Cureton, \u201cSolidarity and social moral rules,\u201d Ethical Theory Moral, vol.15, no.5, pp.691-706, 2012. 10.1007\/s10677-011-9313-8","DOI":"10.1007\/s10677-011-9313-8"},{"key":"20","doi-asserted-by":"crossref","unstructured":"[20] Y. Guo, H. Dibeklioglu, and L.V.D. Maaten, \u201cGraph-based kinship recognition,\u201d Proc. Int. Conf. Pattern Recognition, pp.4287-4292, 2014. 10.1109\/icpr.2014.735","DOI":"10.1109\/ICPR.2014.735"},{"key":"21","doi-asserted-by":"crossref","unstructured":"[21] J. Li, Y. Wong, Q. Zhao, and M.S. Kankanhalli, \u201cDual-glance model for deciphering social relationships,\u201d Proc. IEEE Int. Conf. Computer Vision, pp.2669-2678, 2017. 10.1109\/iccv.2017.289","DOI":"10.1109\/ICCV.2017.289"},{"key":"22","doi-asserted-by":"crossref","unstructured":"[22] W. Pei, T. Baltrusaitis, D.M.J. Tax, and L.P. Morency, \u201cTemporal attention-gated model for robust sequence classification,\u201d Proc. IEEE Int. Conf. Computer Vision, pp.820-829, 2016. 10.1109\/cvpr.2017.94","DOI":"10.1109\/CVPR.2017.94"},{"key":"23","doi-asserted-by":"publisher","unstructured":"[23] N. Ambady and R. Rosenthal, \u201cThin slices of expressive behavior as predictors of interpersonal consequences: A meta-analysis,\u201d Psychol Bull, vol.11, no.2, pp.256-274, 1992. 10.1037\/\/0033-2909.111.2.256","DOI":"10.1037\/\/0033-2909.111.2.256"},{"key":"24","doi-asserted-by":"crossref","unstructured":"[24] H. Yu, L. Gui, M. Madaio, A. Ogan, J. Cassell, and L.-P. Morency, \u201cTemporally selective attention model for social and affective state recognition in multimedia content,\u201d Proc. ACM on Multimedia Conference, pp.1743-1751, 2017. 10.1145\/3123266.3123413","DOI":"10.1145\/3123266.3123413"},{"key":"25","doi-asserted-by":"crossref","unstructured":"[25] P. Vicol, M. Tapaswi, L. Castrej\u00f3n, and S. Fidler, \u201cMovieGraphs: Towards Understanding Human-Centric Situations From Videos,\u201d Proc. IEEE Conf. Computer Vision and Pattern Recognition, pp.8581-8590, 2018. 10.1109\/cvpr.2018.00895","DOI":"10.1109\/CVPR.2018.00895"},{"key":"26","doi-asserted-by":"crossref","unstructured":"[26] J. Xu, T. Yao, Y. Zhang, and T. Mei, \u201cLearning multimodal attention lstm networks for video captioning,\u201d Proc. IEEE Conf. Computer Vision and Pattern Recognition, pp.537-545, 2017. 10.1145\/3123266.3123448","DOI":"10.1145\/3123266.3123448"},{"key":"27","doi-asserted-by":"crossref","unstructured":"[27] L. Baraldi, C. Grana, and R. Cucchiara, \u201cHierarchical boundary-aware neural encoder for video captioning,\u201d Proc. IEEE Conf. Computer Vision and Pattern Recognition, pp.3185-3194, 2017. 10.1109\/cvpr.2017.339","DOI":"10.1109\/CVPR.2017.339"},{"key":"28","unstructured":"[28] T. Mikolov, M. Karafi\u00e1t, L. Burget, J. \u010cernock\u1ef3, and S. Khudanpur, \u201cRecurrent neural network based language model,\u201d Proc. Int. Conf. Speech Communication Association, pp.1045-1048, 2010."},{"key":"29","doi-asserted-by":"crossref","unstructured":"[29] M. Tufano, J. Pantiuchina, C. Watson, G. Bavota, and D. Poshyvanyk, \u201cOn Learning Meaningful Code Changes via Neural Machine Translation,\u201d arXiv preprint arXiv:1901.09102, 2019. 10.1109\/icse.2019.00021","DOI":"10.1109\/ICSE.2019.00021"},{"key":"30","doi-asserted-by":"crossref","unstructured":"[30] J. Xu, T. Yao, Y. Zhang, and T. Mei, \u201cLearning multimodal attention lstm networks for video captioning,\u201d Proc. ACM on Multimedia, pp.537-545, 2017. 10.1145\/3123266.3123448","DOI":"10.1145\/3123266.3123448"},{"key":"31","doi-asserted-by":"publisher","unstructured":"[31] Y. Li, Z. Miao, M. He, Y. Zhang, and H. Li, \u201cDeep Attention Residual Hashing,\u201d IEICE Transactions, vol.E101-A, no.3, pp.654-657, 2018. 10.1587\/transfun.e101.a.654","DOI":"10.1587\/transfun.E101.A.654"},{"key":"32","doi-asserted-by":"publisher","unstructured":"[32] Y. Bin, Y. Yang, F. Shen, N. Xie, H.T. Shen, and X. Li, \u201cDescribing Video With Attention-Based Bidirectional LSTM,\u201d IEEE Transactions on Cybernetics, vol.49, no.7, pp.2631-2641, 2018. 10.1109\/tcyb.2018.2831447","DOI":"10.1109\/TCYB.2018.2831447"},{"key":"33","doi-asserted-by":"publisher","unstructured":"[33] T. Rao, X. Li, H. Zhang, and M. Xu, \u201cMulti-level region-based Convolutional Neural Network for image emotion classification,\u201d Neurocomputing, vol.333, pp.429-439, 2019. 10.1016\/j.neucom.2018.12.053","DOI":"10.1016\/j.neucom.2018.12.053"},{"key":"34","unstructured":"[34] R. Girdhar and D. Ramanan, \u201cAttentional pooling for action recognition,\u201d Proc. Conf. Neural Information Processing Systems, pp.34-45, 2017."},{"key":"35","doi-asserted-by":"crossref","unstructured":"[35] F. Zhu, H. Li, W. Ouyang, N. Yu, and X. Wang, \u201cLearning spatial regularization with image-level supervisions for multi-label image classification,\u201d Proc. IEEE Conf. Computer Vision and Pattern Recognition, pp.2027-2036, 2017. 10.1109\/cvpr.2017.219","DOI":"10.1109\/CVPR.2017.219"},{"key":"36","doi-asserted-by":"crossref","unstructured":"[36] M.C. Phan, A. Sun, Y. Tay, J. Han, and C. Li, \u201cNeupl: Attention-based semantic matching and pair-linking for entity disambiguation,\u201d Proc. Conf. Information and Knowledge Management, pp.1667-1676, 2017. 10.1145\/3132847.3132963","DOI":"10.1145\/3132847.3132963"},{"key":"37","doi-asserted-by":"crossref","unstructured":"[37] D. Tran, L. Bourdev, R. Fergus, L. Torresani, and M. Paluri, \u201cLearning spatiotemporal features with 3d convolutional networks,\u201d Proc. IEEE Conf. Computer Vision and Pattern Recognition, pp.4489-4497, 2015. 10.1109\/iccv.2015.510","DOI":"10.1109\/ICCV.2015.510"},{"key":"38","doi-asserted-by":"crossref","unstructured":"[38] L. Wang, Y. Xiong, Z. Wang, Y. Qiao, D. Lin, X. Tang, and L.V. Gool, \u201cTemporal segment networks: towards good practices for deep action recognition,\u201d Proc. European Conference on Computer Vision, pp.20-36, 2016. 10.1007\/978-3-319-46484-8_2","DOI":"10.1007\/978-3-319-46484-8_2"},{"key":"39","doi-asserted-by":"publisher","unstructured":"[39] N.V. Findler, \u201cShort note on a heuristic search strategy in long-term memory networks,\u201d Inf. Process. Lett, vol.1, no.5, pp.191-196, 1972. 10.1016\/0020-0190(72)90037-3","DOI":"10.1016\/0020-0190(72)90037-3"}],"container-title":["IEICE Transactions on Information and Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E102.D\/12\/E102.D_2019EDP7104\/_pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,12,7]],"date-time":"2019-12-07T03:31:51Z","timestamp":1575689511000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E102.D\/12\/E102.D_2019EDP7104\/_article"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,12,1]]},"references-count":39,"journal-issue":{"issue":"12","published-print":{"date-parts":[[2019]]}},"URL":"https:\/\/doi.org\/10.1587\/transinf.2019edp7104","relation":{},"ISSN":["0916-8532","1745-1361"],"issn-type":[{"value":"0916-8532","type":"print"},{"value":"1745-1361","type":"electronic"}],"subject":[],"published":{"date-parts":[[2019,12,1]]}}}