{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,29]],"date-time":"2025-12-29T18:54:28Z","timestamp":1767034468242,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":68,"publisher":"ACM","license":[{"start":{"date-parts":[[2019,10,14]],"date-time":"2019-10-14T00:00:00Z","timestamp":1571011200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2019,10,14]]},"DOI":"10.1145\/3340555.3353761","type":"proceedings-article","created":{"date-parts":[[2019,10,17]],"date-time":"2019-10-17T12:49:48Z","timestamp":1571316588000},"page":"385-394","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":12,"title":["Improved Visual Focus of Attention Estimation and Prosodic Features for Analyzing Group Interactions"],"prefix":"10.1145","author":[{"given":"Lingyu","family":"Zhang","sequence":"first","affiliation":[{"name":"Rensselaer Polytechnic Institute"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mallory","family":"Morgan","sequence":"additional","affiliation":[{"name":"Rensselaer Polytechnic Institute, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Indrani","family":"Bhattacharya","sequence":"additional","affiliation":[{"name":"Rensselaer Polytechnic Institute, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Michael","family":"Foley","sequence":"additional","affiliation":[{"name":"Northeastern University, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jonas","family":"Braasch","sequence":"additional","affiliation":[{"name":"Rensselaer Polytechnic Institute, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Christoph","family":"Riedl","sequence":"additional","affiliation":[{"name":"Northeastern University, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Brooke","family":"Foucault Welles","sequence":"additional","affiliation":[{"name":"Northeastern University, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Richard J.","family":"Radke","sequence":"additional","affiliation":[{"name":"Rensselaer Polytechnic Institute, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2019,10,14]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"crossref","unstructured":"F. Al\u00edas J.C. Socor\u00f3 and X. Sevillano. 2016. A Review of Physical and Perceptual Feature Extraction Techniques for Speech Music and Environmental Sounds. Appl. Sci. 6 143 (2016).  F. Al\u00edas J.C. Socor\u00f3 and X. Sevillano. 2016. A Review of Physical and Perceptual Feature Extraction Techniques for Speech Music and Environmental Sounds. Appl. Sci. 6 143 (2016).","DOI":"10.3390\/app6050143"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/TSMCB.2008.927274"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2010.69"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/FG.2018.00019"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1111\/1469-7610.00715"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1037\/0021-9010.82.1.62"},{"volume-title":"Proceedings of the 19th ACM International Conference on Multimodal Interaction. ACM, 451\u2013455","author":"Beyan C.","key":"e_1_3_2_1_7_1","unstructured":"C. Beyan , F. Capozzi , C. Becchio , and V. Murino . 2017. Multi-task learning of social psychology assessments and nonverbal features for automatic leadership identification . In Proceedings of the 19th ACM International Conference on Multimodal Interaction. ACM, 451\u2013455 . C. Beyan, F. Capozzi, C. Becchio, and V. Murino. 2017. Multi-task learning of social psychology assessments and nonverbal features for automatic leadership identification. In Proceedings of the 19th ACM International Conference on Multimodal Interaction. ACM, 451\u2013455."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2017.2740062"},{"volume-title":"Proceedings of the 18th ACM International Conference on Multimodal Interaction. ACM, 317\u2013324","author":"Beyan C.","key":"e_1_3_2_1_9_1","unstructured":"C. Beyan , N. Carissimi , F. Capozzi , S. Vascon , M. Bustreo , A. Pierro , C. Becchio , and V. Murino . 2016. Detecting emergent leader in a meeting environment using nonverbal visual features only . In Proceedings of the 18th ACM International Conference on Multimodal Interaction. ACM, 317\u2013324 . C. Beyan, N. Carissimi, F. Capozzi, S. Vascon, M. Bustreo, A. Pierro, C. Becchio, and V. Murino. 2016. Detecting emergent leader in a meeting environment using nonverbal visual features only. In Proceedings of the 18th ACM International Conference on Multimodal Interaction. ACM, 317\u2013324."},{"volume-title":"Proceedings of the 10th ACM Multimedia Systems Conference(MMSys \u201919)","author":"Bhattacharya I.","key":"e_1_3_2_1_10_1","unstructured":"I. Bhattacharya , M. Foley , C. Ku , N. Zhang , T. Zhang , C. Mine , M. Li , H. Ji , C. Riedl , B. Foucault\u00a0Welles , and R.J. Radke . 2019. The Unobtrusive Group Interaction (UGI) Corpus . In Proceedings of the 10th ACM Multimedia Systems Conference(MMSys \u201919) . I. Bhattacharya, M. Foley, C. Ku, N. Zhang, T. Zhang, C. Mine, M. Li, H. Ji, C. Riedl, B. Foucault\u00a0Welles, and R.J. Radke. 2019. The Unobtrusive Group Interaction (UGI) Corpus. In Proceedings of the 10th ACM Multimedia Systems Conference(MMSys \u201919)."},{"volume-title":"Proceedings of the 2018 International Conference on Multimodal Interaction. ACM, 347\u2013355","author":"Bhattacharya I.","key":"e_1_3_2_1_11_1","unstructured":"I. Bhattacharya , M. Foley , N. Zhang , T. Zhang , C. Ku , C. Mine , H. Ji , C. Riedl , B. Foucault\u00a0Welles , and R.J. Radke . 2018. A Multimodal-Sensor-Enabled Room for Unobtrusive Group Meeting Analysis . In Proceedings of the 2018 International Conference on Multimodal Interaction. ACM, 347\u2013355 . I. Bhattacharya, M. Foley, N. Zhang, T. Zhang, C. Ku, C. Mine, H. Ji, C. Riedl, B. Foucault\u00a0Welles, and R.J. Radke. 2018. A Multimodal-Sensor-Enabled Room for Unobtrusive Group Meeting Analysis. In Proceedings of the 2018 International Conference on Multimodal Interaction. ACM, 347\u2013355."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1108\/02621719710174525"},{"key":"e_1_3_2_1_13_1","volume-title":"JSU Report 1003. Joint Speech Research Unit, Ruislip, England.","author":"Bridle J.S.","year":"1974","unstructured":"J.S. Bridle and M.D. Brown . 1974 . An Experimental Automatic Word-Recognition System . JSU Report 1003. Joint Speech Research Unit, Ruislip, England. J.S. Bridle and M.D. Brown. 1974. An Experimental Automatic Word-Recognition System. JSU Report 1003. Joint Speech Research Unit, Ruislip, England."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/MPRV.2010.86"},{"volume-title":"The ISL meeting corpus: The impact of meeting type on speech style. In INTERSPEECH","author":"Burger S.","key":"e_1_3_2_1_15_1","unstructured":"S. Burger , V. MacLaren , and H. Yu . 2002 . The ISL meeting corpus: The impact of meeting type on speech style. In INTERSPEECH . Denver, CO. S. Burger, V. MacLaren, and H. Yu. 2002. The ISL meeting corpus: The impact of meeting type on speech style. In INTERSPEECH. Denver, CO."},{"volume-title":"Proc. Int. Conf. Lang. Resources Evaluation","author":"Campbell N.","key":"e_1_3_2_1_16_1","unstructured":"N. Campbell , T. Sadanobu , M. Imura , N. Iwahashi , S. Noriko , and D. Douxchamps . 2006. A multimedia database of meetings and informal interactions for tracking participant involvement and discourse flow . In Proc. Int. Conf. Lang. Resources Evaluation . Genoa, Italy. N. Campbell, T. Sadanobu, M. Imura, N. Iwahashi, S. Noriko, and D. Douxchamps. 2006. A multimedia database of meetings and informal interactions for tracking participant involvement and discourse flow. In Proc. Int. Conf. Lang. Resources Evaluation. Genoa, Italy."},{"key":"e_1_3_2_1_17_1","volume-title":"The AMI meeting corpus: A pre-announcement. In International Workshop on Machine Learning for Multimodal Interaction. Springer, 28\u201339","author":"Carletta J.","year":"2005","unstructured":"J. Carletta , S. Ashby , S. Bourban , M. Flynn , M. Guillemot , T. Hain , 2005 . The AMI meeting corpus: A pre-announcement. In International Workshop on Machine Learning for Multimodal Interaction. Springer, 28\u201339 . J. Carletta, S. Ashby, S. Bourban, M. Flynn, M. Guillemot, T. Hain, 2005. The AMI meeting corpus: A pre-announcement. In International Workshop on Machine Learning for Multimodal Interaction. Springer, 28\u201339."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"crossref","unstructured":"J.W. Chang T. Sy and J.N. Change. 2012. Team Emotional Intelligence and Performance: Interactive Dynamics between Leaders and Members. Small Group Research 43 1 (2012).  J.W. Chang T. Sy and J.N. Change. 2012. Team Emotional Intelligence and Performance: Interactive Dynamics between Leaders and Members. Small Group Research 43 1 (2012).","DOI":"10.1177\/1046496411415692"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.3115\/1072064.1072067"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1108\/00251740610668897"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1007\/BF00994018"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"crossref","unstructured":"P.L. Cur\u015feu R. Ilies D. Virg\u01ce L. Marticu\u0163oiu and F.A. Sava. 2018. Personality characteristics that are valued in teams: Not always \u201cmore is better\u201d?International Journal of Psychology(2018).  P.L. Cur\u015feu R. Ilies D. Virg\u01ce L. Marticu\u0163oiu and F.A. Sava. 2018. Personality characteristics that are valued in teams: Not always \u201cmore is better\u201d?International Journal of Psychology(2018).","DOI":"10.1002\/ijop.12511"},{"key":"e_1_3_2_1_23_1","unstructured":"A. Darioly and M.S. Mast. 2014. The role of nonverbal behavior for leadership: An integrative review. In Leader Interpersonal and Influence Skills: The Soft Skills of Leadership R.E. Riggio and S.\u00a0Tan (Eds.). Taylor and Francis 73\u2013100.  A. Darioly and M.S. Mast. 2014. The role of nonverbal behavior for leadership: An integrative review. In Leader Interpersonal and Influence Skills: The Soft Skills of Leadership R.E. Riggio and S.\u00a0Tan (Eds.). Taylor and Francis 73\u2013100."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1177\/1046496495264002"},{"key":"e_1_3_2_1_25_1","unstructured":"B.M. DePaulo and H.S. Friedman. 1998. Nonverbal communication. In Handbook of Social Psychology(4 ed.) D.\u00a0Gilbert S.\u00a0Fisker and G.\u00a0Lindzey (Eds.). McGraw Hill Boston MA 3\u201340.  B.M. DePaulo and H.S. Friedman. 1998. Nonverbal communication. In Handbook of Social Psychology(4 ed.) D.\u00a0Gilbert S.\u00a0Fisker and G.\u00a0Lindzey (Eds.). McGraw Hill Boston MA 3\u201340."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1016\/S1746-9791(06)02002-5"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2015.2501920"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1111\/j.1744-6570.2003.tb00754.x"},{"volume-title":"11th Annual Conference of the International Speech (InterSpeech)","author":"Ghaemmaghami H.","key":"e_1_3_2_1_29_1","unstructured":"H. Ghaemmaghami , B. Baker , R. Vogt , and S. Sridharan . 2010. Noise robust voice activity detection using features extracted from the time-domain autocorrelation function . In 11th Annual Conference of the International Speech (InterSpeech) . Makuhari, Japan, 3118\u20133121. H. Ghaemmaghami, B. Baker, R. Vogt, and S. Sridharan. 2010. Noise robust voice activity detection using features extracted from the time-domain autocorrelation function. In 11th Annual Conference of the International Speech (InterSpeech). Makuhari, Japan, 3118\u20133121."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2013.2295918"},{"key":"e_1_3_2_1_31_1","volume-title":"Meetings: Factors that affect group interaction and performance. Proceedings of the Association of Researchers in Construction Management","author":"Gorse C.","year":"2006","unstructured":"C. Gorse , I. McKinney , A. Shepherd , and P. Whitehead . 2006 . Meetings: Factors that affect group interaction and performance. Proceedings of the Association of Researchers in Construction Management ( 2006 ), 4\u20136. C. Gorse, I. McKinney, A. Shepherd, and P. Whitehead. 2006. Meetings: Factors that affect group interaction and performance. Proceedings of the Association of Researchers in Construction Management (2006), 4\u20136."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASSP.1974.1162572"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1037\/0033-2909.131.6.898"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1177\/001872677002300404"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1111\/j.1467-839X.2004.00149.x"},{"volume-title":"Detecting Harmonic Change in Musical Audio. In 1st ACM Workshop on Audio and Music Computing Multimedia. ACM","author":"Harte C.","key":"e_1_3_2_1_36_1","unstructured":"C. Harte , M. Sandler , and M. Gasser . 2006 . Detecting Harmonic Change in Musical Audio. In 1st ACM Workshop on Audio and Music Computing Multimedia. ACM , Santa Barbara, CA, 21\u201326. C. Harte, M. Sandler, and M. Gasser. 2006. Detecting Harmonic Change in Musical Audio. In 1st ACM Workshop on Audio and Music Computing Multimedia. ACM, Santa Barbara, CA, 21\u201326."},{"volume-title":"2011 International Conference on Computer Vision. IEEE, 383\u2013390","author":"J.","key":"e_1_3_2_1_37_1","unstructured":"J. A Hesch and S.I Roumeliotis. 2011. A direct least-squares (DLS) method for PnP . In 2011 International Conference on Computer Vision. IEEE, 383\u2013390 . J.A Hesch and S.I Roumeliotis. 2011. A direct least-squares (DLS) method for PnP. In 2011 International Conference on Computer Vision. IEEE, 383\u2013390."},{"key":"e_1_3_2_1_38_1","volume-title":"Ternausnet: U-net with VGG11 encoder pre-trained on Imagenet for image segmentation. arXiv preprint arXiv:1801.05746(2018).","author":"Iglovikov V.","year":"2018","unstructured":"V. Iglovikov and A. Shvets . 2018 . Ternausnet: U-net with VGG11 encoder pre-trained on Imagenet for image segmentation. arXiv preprint arXiv:1801.05746(2018). V. Iglovikov and A. Shvets. 2018. Ternausnet: U-net with VGG11 encoder pre-trained on Imagenet for image segmentation. arXiv preprint arXiv:1801.05746(2018)."},{"key":"e_1_3_2_1_39_1","volume-title":"The ICSI meeting corpus. In Int. Conf. Acoust., Speech, and Signal Process.","author":"Janin A.","year":"2003","unstructured":"A. Janin , D. Baron , J. Edwards , D. Ellis , D. Gelbart , N. Morgan , 2003 . The ICSI meeting corpus. In Int. Conf. Acoust., Speech, and Signal Process. A. Janin, D. Baron, J. Edwards, D. Ellis, D. Gelbart, N. Morgan, 2003. The ICSI meeting corpus. In Int. Conf. Acoust., Speech, and Signal Process."},{"volume-title":"Proceedings of the 14th ACM International Conference on Multimodal Interaction. ACM, 433\u2013440","author":"Jayagopi D.","key":"e_1_3_2_1_40_1","unstructured":"D. Jayagopi , D. Sanchez-Cortes , K. Otsuka , J. Yamato , and D. Gatica-Perez . 2012. Linking speaking and looking behavior patterns with group composition, perception, and performance . In Proceedings of the 14th ACM International Conference on Multimodal Interaction. ACM, 433\u2013440 . D. Jayagopi, D. Sanchez-Cortes, K. Otsuka, J. Yamato, and D. Gatica-Perez. 2012. Linking speaking and looking behavior patterns with group composition, perception, and performance. In Proceedings of the 14th ACM International Conference on Multimodal Interaction. ACM, 433\u2013440."},{"volume-title":"International Conference on Multimedia and Expo. 113\u2013116","author":"Jiang D.","key":"e_1_3_2_1_41_1","unstructured":"D. Jiang , L. Lu , H. Zhang , J. Tao , and L. Cai . 2002. Music type classification by spectral contrast feature . In International Conference on Multimedia and Expo. 113\u2013116 . D. Jiang, L. Lu, H. Zhang, J. Tao, and L. Cai. 2002. Music type classification by spectral contrast feature. In International Conference on Multimedia and Expo. 113\u2013116."},{"key":"e_1_3_2_1_42_1","unstructured":"O.P. John and S. Srivastava. 1999. The Big Five trait taxonomy: History measurement and theoretical perspectives. In Handbook Personality: Theory and Research (2 ed.) L.A. Pervin and O.P. John (Eds.). McGraw Hill Boston MA 102\u2013138.  O.P. John and S. Srivastava. 1999. The Big Five trait taxonomy: History measurement and theoretical perspectives. In Handbook Personality: Theory and Research (2 ed.) L.A. Pervin and O.P. John (Eds.). McGraw Hill Boston MA 102\u2013138."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1016\/S0923-4748(97)00010-6"},{"key":"e_1_3_2_1_44_1","first-page":"57","article-title":"Audio Content Classification Method Research Based on Two-step","volume":"5","author":"Liang S.","year":"2014","unstructured":"S. Liang and X. Fan . 2014 . Audio Content Classification Method Research Based on Two-step Strategy. Int. J. Adv. Comput. Sci. Appl. 5 (2014), 57 \u2013 62 . S. Liang and X. Fan. 2014. Audio Content Classification Method Research Based on Two-step Strategy. Int. J. Adv. Comput. Sci. Appl. 5 (2014), 57\u201362.","journal-title":"Strategy. Int. J. Adv. Comput. Sci. Appl."},{"volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. 3431\u20133440","author":"Long J.","key":"e_1_3_2_1_45_1","unstructured":"J. Long , E. Shelhamer , and T. Darrell . 2015. Fully convolutional networks for semantic segmentation . In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. 3431\u20133440 . J. Long, E. Shelhamer, and T. Darrell. 2015. Fully convolutional networks for semantic segmentation. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. 3431\u20133440."},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1016\/0030-5073(84)90043-6"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2017.2782819"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.socnet.2012.04.002"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2005.49"},{"volume-title":"Estimating Visual Focus of Attention in Multiparty Meetings using Deep Convolutional Neural Networks. In Proceedings of the 2018 on International Conference on Multimodal Interaction. ACM, 191\u2013199","author":"Otsuka K.","key":"e_1_3_2_1_50_1","unstructured":"K. Otsuka , K. Kasuga , and M. K\u00f6hler . 2018 . Estimating Visual Focus of Attention in Multiparty Meetings using Deep Convolutional Neural Networks. In Proceedings of the 2018 on International Conference on Multimodal Interaction. ACM, 191\u2013199 . K. Otsuka, K. Kasuga, and M. K\u00f6hler. 2018. Estimating Visual Focus of Attention in Multiparty Meetings using Deep Convolutional Neural Networks. In Proceedings of the 2018 on International Conference on Multimodal Interaction. ACM, 191\u2013199."},{"volume-title":"Proc. Int. Conf. Multimodal Interfaces. ACM","author":"Otsuka K.","key":"e_1_3_2_1_51_1","unstructured":"K. Otsuka , H. Sawada , and J. Yamato . 2007. Automatic inference of cross-modal nonverbal interactions in multiparty conversations: Who responds to whom, when, and how? From gaze, head gestures, and utterances . In Proc. Int. Conf. Multimodal Interfaces. ACM , Aichi, Japan. K. Otsuka, H. Sawada, and J. Yamato. 2007. Automatic inference of cross-modal nonverbal interactions in multiparty conversations: Who responds to whom, when, and how? From gaze, head gestures, and utterances. In Proc. Int. Conf. Multimodal Interfaces. ACM, Aichi, Japan."},{"volume-title":"Proceedings of the 7th International Conference on Multimodal Interfaces. ACM, 191\u2013198","author":"Otsuka K.","key":"e_1_3_2_1_52_1","unstructured":"K. Otsuka , Y. Takemae , and J. Yamato . 2005. A probabilistic inference of multiparty-conversation structure based on Markov-switching models of gaze patterns, head directions, and utterances . In Proceedings of the 7th International Conference on Multimodal Interfaces. ACM, 191\u2013198 . K. Otsuka, Y. Takemae, and J. Yamato. 2005. A probabilistic inference of multiparty-conversation structure based on Markov-switching models of gaze patterns, head directions, and utterances. In Proceedings of the 7th International Conference on Multimodal Interfaces. ACM, 191\u2013198."},{"volume-title":"Proc. Int. Conf. Multimedia and Expo.IEEE","author":"Otsuka K.","key":"e_1_3_2_1_53_1","unstructured":"K. Otsuka , J. Yamato , Y. Takemae , and H. Murase . 2006. Conversation scene analysis with dynamic Bayesian Network based on visual head tracking . In Proc. Int. Conf. Multimedia and Expo.IEEE , Toronto, ON, Canada. K. Otsuka, J. Yamato, Y. Takemae, and H. Murase. 2006. Conversation scene analysis with dynamic Bayesian Network based on visual head tracking. In Proc. Int. Conf. Multimedia and Expo.IEEE, Toronto, ON, Canada."},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1177\/104649640103200104"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jrp.2006.02.001"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1177\/002194368101800303"},{"key":"e_1_3_2_1_57_1","volume-title":"U-net: Convolutional networks for biomedical image segmentation. In International Conference on Medical Image Computing and Computer-Assisted Intervention","author":"Ronneberger O.","year":"2015","unstructured":"O. Ronneberger , P. Fischer , and T. Brox . 2015 . U-net: Convolutional networks for biomedical image segmentation. In International Conference on Medical Image Computing and Computer-Assisted Intervention . Springer , 234\u2013241. O. Ronneberger, P. Fischer, and T. Brox. 2015. U-net: Convolutional networks for biomedical image segmentation. In International Conference on Medical Image Computing and Computer-Assisted Intervention. Springer, 234\u2013241."},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"crossref","unstructured":"D.E. Rumelhart G.E. Hinton and R.J. Williams. 1985. Learning internal representations by error propagation. Technical Report. California Univ San Diego La Jolla Inst for Cognitive Science.  D.E. Rumelhart G.E. Hinton and R.J. Williams. 1985. Learning internal representations by error propagation. Technical Report. California Univ San Diego La Jolla Inst for Cognitive Science.","DOI":"10.21236\/ADA164453"},{"volume-title":"Workshop Multimodal Corpora Mach. Learning: Taking Stock and Road Mapping the Future","author":"Sanchez-Cortes D.","key":"e_1_3_2_1_59_1","unstructured":"D. Sanchez-Cortes , O. Aran , and D. Gatica-Perez . 2011. An audio visual corpus for emergent leader analysis . In Workshop Multimodal Corpora Mach. Learning: Taking Stock and Road Mapping the Future . Alicante, Spain. D. Sanchez-Cortes, O. Aran, and D. Gatica-Perez. 2011. An audio visual corpus for emergent leader analysis. In Workshop Multimodal Corpora Mach. Learning: Taking Stock and Road Mapping the Future. Alicante, Spain."},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2011.2181941"},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1996.543290"},{"volume-title":"Proceedings of the IEEE International Conference on Acoustics, Speech, and Signal Processing. 1331\u20131334","author":"Scheirer E.","key":"e_1_3_2_1_62_1","unstructured":"E. Scheirer and M. Slaney . 1997. Construction and evaluation of a robust multifeature speech\/music discriminator . In Proceedings of the IEEE International Conference on Acoustics, Speech, and Signal Processing. 1331\u20131334 . E. Scheirer and M. Slaney. 1997. Construction and evaluation of a robust multifeature speech\/music discriminator. In Proceedings of the IEEE International Conference on Acoustics, Speech, and Signal Processing. 1331\u20131334."},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jvoice.2014.09.016"},{"volume-title":"Model-Based Classification of Speech Audio. Master\u2019s thesis","author":"Thoman C.","key":"e_1_3_2_1_64_1","unstructured":"C. Thoman . 2009. Model-Based Classification of Speech Audio. Master\u2019s thesis . Florida Atlantic University, Florida , USA. C. Thoman. 2009. Model-Based Classification of Speech Audio. Master\u2019s thesis. Florida Atlantic University, Florida, USA."},{"key":"e_1_3_2_1_65_1","volume-title":"Proceedings of the 4th International Society for Music Information Retrieval Conference","author":"Wang A.L.C.","year":"2003","unstructured":"A.L.C. Wang . 2003 . An industrial-strength audio search algorithm . In Proceedings of the 4th International Society for Music Information Retrieval Conference . Baltimore, MD, 7\u201313. A.L.C. Wang. 2003. An industrial-strength audio search algorithm. In Proceedings of the 4th International Society for Music Information Retrieval Conference. Baltimore, MD, 7\u201313."},{"volume-title":"Proceedings of the 10th International Society for Music Information Retrieval Conference","author":"Wang F.","key":"e_1_3_2_1_66_1","unstructured":"F. Wang , X. Wang , B. Shao , T. Li , and M. Ogihara . 2009. Tag Integrated Multi-Label Music Style Classification with Hypergraph . In Proceedings of the 10th International Society for Music Information Retrieval Conference . Kobe, Japan, 363\u2013368. F. Wang, X. Wang, B. Shao, T. Li, and M. Ogihara. 2009. Tag Integrated Multi-Label Music Style Classification with Hypergraph. In Proceedings of the 10th International Society for Music Information Retrieval Conference. Kobe, Japan, 363\u2013368."},{"key":"e_1_3_2_1_67_1","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2016.2603342"},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"publisher","DOI":"10.1145\/319463.319471"}],"event":{"name":"ICMI '19: INTERNATIONAL CONFERENCE ON MULTIMODAL INTERACTION","acronym":"ICMI '19","location":"Suzhou China"},"container-title":["2019 International Conference on Multimodal Interaction"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3340555.3353761","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3340555.3353761","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T23:13:28Z","timestamp":1750202008000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3340555.3353761"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,10,14]]},"references-count":68,"alternative-id":["10.1145\/3340555.3353761","10.1145\/3340555"],"URL":"https:\/\/doi.org\/10.1145\/3340555.3353761","relation":{},"subject":[],"published":{"date-parts":[[2019,10,14]]},"assertion":[{"value":"2019-10-14","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}