{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2023,10,8]],"date-time":"2023-10-08T18:15:20Z","timestamp":1696788920712},"reference-count":47,"publisher":"Institute of Electronics, Information and Communications Engineers (IEICE)","issue":"10","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEICE Trans. Inf. &amp; Syst."],"published-print":{"date-parts":[[2023,10,1]]},"DOI":"10.1587\/transinf.2023pcp0008","type":"journal-article","created":{"date-parts":[[2023,9,30]],"date-time":"2023-09-30T23:01:25Z","timestamp":1696114885000},"page":"1638-1649","source":"Crossref","is-referenced-by-count":0,"title":["Social Relation Atmosphere Recognition with Relevant Visual Concepts"],"prefix":"10.1587","volume":"E106.D","author":[{"given":"Ying","family":"JI","sequence":"first","affiliation":[{"name":"Graduate School of Informatics, Nagoya University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yu","family":"WANG","sequence":"additional","affiliation":[{"name":"Center for Information and Communication Technology, Hitotsubashi University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kensaku","family":"MORI","sequence":"additional","affiliation":[{"name":"Graduate School of Informatics, Nagoya University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jien","family":"KATO","sequence":"additional","affiliation":[{"name":"College of Information Science and Engineering, Ritsumeikan University"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"532","reference":[{"key":"1","doi-asserted-by":"crossref","unstructured":"[1] Q. Sun, B. Schiele, and M. Fritz, \u201cA domain based approach to social relation recognition,\u201d Proc. IEEE\/CVF conference on computer vision and pattern recognition, pp.435-444, July 2017. 10.1109\/cvpr.2017.54","DOI":"10.1109\/CVPR.2017.54"},{"key":"2","doi-asserted-by":"publisher","unstructured":"[2] D.B. Bugental, \u201cAcquisition of the algorithms of social life: a domain-based approach.,\u201d Psychological bulletin, vol.126, no.2, pp.187-219, 2000. 10.1037\/0033-2909.126.2.187","DOI":"10.1037\/0033-2909.126.2.187"},{"key":"3","doi-asserted-by":"crossref","unstructured":"[3] A. Goel, K.T. Ma, and C. Tan, \u201cAn end-to-end network for generating social relationship graphs,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.11186-11195, 2019. 10.1109\/cvpr.2019.01144","DOI":"10.1109\/CVPR.2019.01144"},{"key":"4","doi-asserted-by":"publisher","unstructured":"[4] M. Wang, X. Du, X. Shu, X. Wang, and J. Tang, \u201cDeep supervised feature selection for social relationship recognition,\u201d Pattern Recognition Letters, vol.138, pp.410-416, 2020. 10.1016\/j.patrec.2020.08.005","DOI":"10.1016\/j.patrec.2020.08.005"},{"key":"5","doi-asserted-by":"crossref","unstructured":"[5] X. Liu, W. Liu, M. Zhang, J. Chen, L. Gao, C. Yan, and T. Mei, \u201cSocial relation recognition from videos via multi-scale spatial-temporal reasoning,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.3561-3569, 2019.","DOI":"10.1109\/CVPR.2019.00368"},{"key":"6","doi-asserted-by":"crossref","unstructured":"[6] J. Lv, W. Liu, L. Zhou, B. Wu, and H. Ma, \u201cMulti-stream fusion model for social relation recognition from videos,\u201d International Conference on Multimedia Modeling, vol.10704, pp.355-368, Springer, 2018. 10.1007\/978-3-319-73603-7_29","DOI":"10.1007\/978-3-319-73603-7_29"},{"key":"7","doi-asserted-by":"crossref","unstructured":"[7] M.J. Mar\u00edn-Jim\u00e9nez, V. Kalogeiton, P. Medina-Suarez, and A. Zisserman, \u201cLaeo-net: revisiting people looking at each other in videos,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.3472-3480, 2019.","DOI":"10.1109\/CVPR.2019.00359"},{"key":"8","doi-asserted-by":"crossref","unstructured":"[8] E. Chong, Y. Wang, N. Ruiz, and J.M. Rehg, \u201cDetecting attended visual targets in video,\u201d Proc. IEEE\/CVF conference on computer vision and pattern recognition, pp.5395-5405, 2020.","DOI":"10.1109\/CVPR42600.2020.00544"},{"key":"9","doi-asserted-by":"crossref","unstructured":"[9] Y. Fang, J. Tang, W. Shen, W. Shen, X. Gu, L. Song, and G. Zhai, \u201cDual attention guided gaze target detection in the wild,\u201d Proc. IEEE\/CVF conference on computer vision and pattern recognition, pp.11385-11394, 2021.","DOI":"10.1109\/CVPR46437.2021.01123"},{"key":"10","doi-asserted-by":"crossref","unstructured":"[10] R. Kothari, S. De Mello, U. Iqbal, W. Byeon, S. Park, and J. Kautz, \u201cWeakly-supervised physically unconstrained gaze estimation,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.9975-9984, 2021.","DOI":"10.1109\/CVPR46437.2021.00985"},{"key":"11","doi-asserted-by":"crossref","unstructured":"[11] J. Lee, S. Kim, S. Kim, J. Park, and K. Sohn, \u201cContext-aware emotion recognition networks,\u201d Proc. IEEE\/CVF international conference on computer vision, pp.10142-10151, 2019.","DOI":"10.1109\/ICCV.2019.01024"},{"key":"12","doi-asserted-by":"crossref","unstructured":"[12] T. Mittal, P. Guhan, U. Bhattacharya, R. Chandra, A. Bera, and D. Manocha, \u201cEmoticon: Context-aware multimodal emotion recognition using frege&apos;s principle,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.14222-14231, 2020.","DOI":"10.1109\/CVPR42600.2020.01424"},{"key":"13","doi-asserted-by":"publisher","unstructured":"[13] Y. Li, J. Zeng, S. Shan, and X. Chen, \u201cOcclusion aware facial expression recognition using cnn with attention mechanism,\u201d IEEE Trans. Image Process., vol.28, no.5, pp.2439-2450, 2018. 10.1109\/tip.2018.2886767","DOI":"10.1109\/TIP.2018.2886767"},{"key":"14","doi-asserted-by":"crossref","unstructured":"[14] J. Wu, L. Wang, L. Wang, J. Guo, and G. Wu, \u201cLearning actor relation graphs for group activity recognition,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.9956-9966, 2019.","DOI":"10.1109\/CVPR.2019.01020"},{"key":"15","doi-asserted-by":"crossref","unstructured":"[15] K. Gavrilyuk, R. Sanford, M. Javan, and C.G.M. Snoek, \u201cActor-transformers for group activity recognition,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.836-845, 2020. 10.1109\/cvpr42600.2020.00092","DOI":"10.1109\/CVPR42600.2020.00092"},{"key":"16","doi-asserted-by":"crossref","unstructured":"[16] S. Li, Q. Cao, L. Liu, K. Yang, S. Liu, J. Hou, and S. Yi, \u201cGroupformer: Group activity recognition with clustered spatial-temporal transformer,\u201d Proc. IEEE\/CVF International Conference on Computer Vision, pp.13648-13657, 2021.","DOI":"10.1109\/ICCV48922.2021.01341"},{"key":"17","doi-asserted-by":"crossref","unstructured":"[17] K. He, G. Gkioxari, P. Doll\u00e1r, and R. Girshick, \u201cMask r-cnn,\u201d Proc. IEEE international conference on computer vision, pp.2980-2988, 2017.","DOI":"10.1109\/ICCV.2017.322"},{"key":"18","doi-asserted-by":"crossref","unstructured":"[18] P. Vicol, M. Tapaswi, L. Castrejon, and S. Fidler, \u201cMoviegraphs: Towards understanding human-centric situations from videos,\u201d Proc. IEEE conference on computer vision and pattern recognition, pp.8581-8590, 2018. 10.1109\/cvpr.2018.00895","DOI":"10.1109\/CVPR.2018.00895"},{"key":"19","doi-asserted-by":"publisher","unstructured":"[19] D.J. Kiesler, \u201cThe 1982 interpersonal circle: A taxonomy for complementarity in human transactions.,\u201d Psychological review, vol.90, no.3, p.185-214, 1983. 10.1037\/0033-295x.90.3.185","DOI":"10.1037\/0033-295X.90.3.185"},{"key":"20","doi-asserted-by":"publisher","unstructured":"[20] D.Y.F. Ho, \u201cInterpersonal relationships and relationship dominance: An analysis based on methodological relationism,\u201d Asian Journal of Social Psychology, vol.1, no.1, pp.1-16, 1998. 10.1111\/1467-839x.00002","DOI":"10.1111\/1467-839X.00002"},{"key":"21","doi-asserted-by":"publisher","unstructured":"[21] K. Simonyan and A. Zisserman, \u201cTwo-stream convolutional networks for action recognition in videos,\u201d Advances in neural information processing systems, vol.27, no.10, 2014. 10.3837\/tiis.2021.10.011","DOI":"10.3837\/tiis.2021.10.011"},{"key":"22","doi-asserted-by":"publisher","unstructured":"[22] W. Song, S. Li, L. Fang, and T. Lu, \u201cHyperspectral image classification with deep feature fusion network,\u201d IEEE Trans. Geosci. Remote Sens., vol.56, no.6, pp.3173-3184, 2018. 10.1109\/tgrs.2018.2794326","DOI":"10.1109\/TGRS.2018.2794326"},{"key":"23","doi-asserted-by":"publisher","unstructured":"[23] K. Jiang, Z. Wang, P. Yi, G. Wang, K. Gu, and J. Jiang, \u201cAtmfn: Adaptive-threshold-based multi-model fusion network for compressed face hallucination,\u201d IEEE Trans. Multimedia, vol.22, no.10, pp.2734-2747, 2019. 10.1109\/tmm.2019.2960586","DOI":"10.1109\/TMM.2019.2960586"},{"key":"24","doi-asserted-by":"publisher","unstructured":"[24] G. Xu, W. Li, and J. Liu, \u201cA social emotion classification approach using multi-model fusion,\u201d Future Generation Computer Systems, vol.102, pp.347-356, 2020. 10.1016\/j.future.2019.07.007","DOI":"10.1016\/j.future.2019.07.007"},{"key":"25","doi-asserted-by":"crossref","unstructured":"[25] S. Bakkali, Z. Ming, M. Coustaty, and M. Rusi\u00f1ol, \u201cVisual and textual deep feature fusion for document image classification,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshops, pp.2394-2403, 2020.","DOI":"10.1109\/CVPRW50498.2020.00289"},{"key":"26","doi-asserted-by":"crossref","unstructured":"[26] S. Wu, Y.-C. Chen, X. Li, A.-C. Wu, J.-J. You, and W.-S. Zheng, \u201cAn enhanced deep feature representation for person re-identification,\u201d 2016 IEEE winter conference on applications of computer vision (WACV), pp.1-8, IEEE, 2016. 10.1109\/wacv.2016.7477681","DOI":"10.1109\/WACV.2016.7477681"},{"key":"27","doi-asserted-by":"crossref","unstructured":"[27] T. Hartley, K. Sidorov, C. Willis, and D. Marshall, \u201cSwag: Superpixels weighted by average gradients for explanations of cnns,\u201d Proc. IEEE\/CVF Winter Conference on Applications of Computer Vision, pp.423-432, 2021. 10.1109\/wacv48630.2021.00047","DOI":"10.1109\/WACV48630.2021.00047"},{"key":"28","doi-asserted-by":"crossref","unstructured":"[28] M.T. Ribeiro, S. Singh, and C. Guestrin, \u201c\u201cwhy should i trust you?\u201d: explaining the predictions of any classifier,\u201d Proc. 22nd ACM SIGKDD international conference on knowledge discovery and data mining, pp.1135-1144, 2016. 10.1145\/2939672.2939778","DOI":"10.1145\/2939672.2939778"},{"key":"29","doi-asserted-by":"publisher","unstructured":"[29] Z. Chen, Y. Bei, and C. Rudin, \u201cConcept whitening for interpretable image recognition,\u201d Nature Machine Intelligence, vol.2, no.12, pp.772-782, 2020. 10.1038\/s42256-020-00265-z","DOI":"10.1038\/s42256-020-00265-z"},{"key":"30","doi-asserted-by":"crossref","unstructured":"[30] J. Wagner, J.M. K\u00f6hler, T. Gindele, L. Hetzel, J.T. Wiedemer, and S. Behnke, \u201cInterpretable and fine-grained visual explanations for convolutional neural networks,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.9089-9099, 2019.","DOI":"10.1109\/CVPR.2019.00931"},{"key":"31","unstructured":"[31] V. Petsiuk, A. Das, and K. Saenko, \u201cRise: Randomized input sampling for explanation of black-box models,\u201d arXiv preprint arXiv:1806.07421, 2018."},{"key":"32","doi-asserted-by":"crossref","unstructured":"[32] B. Zhou, A. Khosla, A. Lapedriza, A. Oliva, and A. Torralba, \u201cLearning deep features for discriminative localization,\u201d Proc. IEEE conference on computer vision and pattern recognition, pp.2921-2929, 2016. 10.1109\/cvpr.2016.319","DOI":"10.1109\/CVPR.2016.319"},{"key":"33","unstructured":"[33] B. Kim, M. Wattenberg, J. Gilmer, C. Cai, J. Wexler, F. Viegas, and R. Sayres, \u201cInterpretability beyond feature attribution: Quantitative testing with concept activation vectors (tcav),\u201d International conference on machine learning, pp.2668-2677, PMLR, 2018."},{"key":"34","unstructured":"[34] A. Ghorbani, J. Wexler, J. Zou, and B. Kim, \u201cTowards automatic concept-based explanations,\u201d arXiv preprint arXiv:1902.03129, 2019."},{"key":"35","doi-asserted-by":"crossref","unstructured":"[35] Y. Ge, Y. Xiao, Z. Xu, M. Zheng, S. Karanam, T. Chen, L. Itti, and Z. Wu, \u201cA peek into the reasoning of neural networks: Interpreting with structural visual concepts,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.2195-2204, 2021. 10.1109\/cvpr46437.2021.00223","DOI":"10.1109\/CVPR46437.2021.00223"},{"key":"36","doi-asserted-by":"crossref","unstructured":"[36] S. Ebrahimi Kahou, V. Michalski, K. Konda, R. Memisevic, and C. Pal, \u201cRecurrent neural networks for emotion recognition in video,\u201d Proc. 2015 ACM on international conference on multimodal interaction, pp.467-474, 2015. 10.1145\/2818346.2830596","DOI":"10.1145\/2818346.2830596"},{"key":"37","doi-asserted-by":"crossref","unstructured":"[37] Y. Fan, X. Lu, D. Li, and Y. Liu, \u201cVideo-based emotion recognition using cnn-rnn and c3d hybrid networks,\u201d Proc. 18th ACM international conference on multimodal interaction, pp.445-450, 2016. 10.1145\/2993148.2997632","DOI":"10.1145\/2993148.2997632"},{"key":"38","doi-asserted-by":"crossref","unstructured":"[38] S.E. Kahou, C. Pal, X. Bouthillier, P. Froumenty, \u00c7. G\u00fcl\u00e7ehre, R. Memisevic, P. Vincent, A. Courville, Y. Bengio, R.C. Ferrari, M. Mirza, S. Jean, P.-L. Carrier, Y. Dauphin, N. Boulanger-Lewandowski, A. Aggarwal, J. Zumer, P. Lamblin, J.-P. Raymond, G. Desjardins, R. Pascanu, D. Warde-Farley, A. Torabi, A. Sharma, E. Bengio, M. C\u00f4t\u00e9, K.R. Konda, and Z. Wu, \u201cCombining modality specific deep neural networks for emotion recognition in video,\u201d Proc. 15th ACM on International conference on multimodal interaction, pp.543-550, 2013. 10.1145\/2522848.2531745","DOI":"10.1145\/2522848.2531745"},{"key":"39","doi-asserted-by":"crossref","unstructured":"[39] X. Gao, Y. Zhao, J. Zhang, and L. Cai, \u201cPairwise emotional relationship recognition in drama videos: Dataset and benchmark,\u201d Proc. 29th ACM International Conference on Multimedia, pp.3380-3389, 2021. 10.1145\/3474085.3475493","DOI":"10.1145\/3474085.3475493"},{"key":"40","doi-asserted-by":"publisher","unstructured":"[40] K. Toyoda, Y. Miyakoshi, R. Yamanishi, and S. Kato, \u201cDialogue mood estimation focusing on intervals of utterance state,\u201d Transactions of the Japanese Society for Artificial Intelligence, vol.27, no.2, pp.16-21, 2012. 10.1527\/tjsai.27.16","DOI":"10.1527\/tjsai.27.16"},{"key":"41","doi-asserted-by":"publisher","unstructured":"[41] R. Achanta, A. Shaji, K. Smith, A. Lucchi, P. Fua, and S. S\u00fcsstrunk, \u201cSlic superpixels compared to state-of-the-art superpixel methods,\u201d IEEE Trans. Pattern Anal. Mach. Intell., vol.34, no.11, pp.2274-2282, 2012. 10.1109\/tpami.2012.120","DOI":"10.1109\/TPAMI.2012.120"},{"key":"42","doi-asserted-by":"crossref","unstructured":"[42] D. Tran, L. Bourdev, R. Fergus, L. Torresani, and M. Paluri, \u201cLearning spatiotemporal features with 3d convolutional networks,\u201d Proc. IEEE international conference on computer vision, pp.4489-4497, 2015. 10.1109\/iccv.2015.510","DOI":"10.1109\/ICCV.2015.510"},{"key":"43","doi-asserted-by":"crossref","unstructured":"[43] D. Tran, H. Wang, L. Torresani, J. Ray, Y. LeCun, and M. Paluri, \u201cA closer look at spatiotemporal convolutions for action recognition,\u201d Proc. IEEE conference on Computer Vision and Pattern Recognition, pp.6450-6459, 2018. 10.1109\/cvpr.2018.00675","DOI":"10.1109\/CVPR.2018.00675"},{"key":"44","doi-asserted-by":"crossref","unstructured":"[44] J. Carreira and A. Zisserman, \u201cQuo vadis, action recognition? a new model and the kinetics dataset,\u201d Proc. IEEE Conference on Computer Vision and Pattern Recognition, pp.4724-4733, 2017.","DOI":"10.1109\/CVPR.2017.502"},{"key":"45","doi-asserted-by":"crossref","unstructured":"[45] H. Kuehne, H. Jhuang, E. Garrote, T. Poggio, and T. Serre, \u201cHMDB: a large video database for human motion recognition,\u201d Proc. International Conference on Computer Vision (ICCV), pp.2556-2563, 2011. 10.1109\/iccv.2011.6126543","DOI":"10.1109\/ICCV.2011.6126543"},{"key":"46","doi-asserted-by":"crossref","unstructured":"[46] L.D. Nguyen, D. Lin, Z. Lin, and J. Cao, \u201cDeep cnns for microscopic image classification by exploiting transfer learning and feature concatenation,\u201d 2018 IEEE international symposium on circuits and systems (ISCAS), pp.1-5, IEEE, 2018. 10.1109\/iscas.2018.8351550","DOI":"10.1109\/ISCAS.2018.8351550"},{"key":"47","doi-asserted-by":"publisher","unstructured":"[47] Z. Liu, W. Hou, J. Zhang, C. Cao, and B. Wu, \u201cA multimodal approach for multiple-relation extraction in videos,\u201d Multimedia Tools and Applications, vol.81, no.4, pp.4909-4934, 2022. 10.1007\/s11042-021-11466-y","DOI":"10.1007\/s11042-021-11466-y"}],"container-title":["IEICE Transactions on Information and Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E106.D\/10\/E106.D_2023PCP0008\/_pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,10,7]],"date-time":"2023-10-07T04:24:23Z","timestamp":1696652663000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E106.D\/10\/E106.D_2023PCP0008\/_article"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,1]]},"references-count":47,"journal-issue":{"issue":"10","published-print":{"date-parts":[[2023]]}},"URL":"https:\/\/doi.org\/10.1587\/transinf.2023pcp0008","relation":{},"ISSN":["0916-8532","1745-1361"],"issn-type":[{"value":"0916-8532","type":"print"},{"value":"1745-1361","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,10,1]]},"article-number":"2023PCP0008"}}