{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,10]],"date-time":"2026-01-10T08:00:07Z","timestamp":1768032007395,"version":"3.49.0"},"reference-count":44,"publisher":"Institute of Electronics, Information and Communications Engineers (IEICE)","issue":"6","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEICE Trans. Inf. &amp; Syst."],"published-print":{"date-parts":[[2025,6,1]]},"DOI":"10.1587\/transinf.2024hcp0002","type":"journal-article","created":{"date-parts":[[2024,11,12]],"date-time":"2024-11-12T22:11:20Z","timestamp":1731449480000},"page":"445-453","source":"Crossref","is-referenced-by-count":1,"title":["Multimodal Voice Activity Projection for Turn-Taking and Effects on Speaker Adaptation"],"prefix":"10.1587","volume":"E108.D","author":[{"given":"Kazuyo","family":"ONISHI","sequence":"first","affiliation":[{"name":"Nara Institute of Science and Technology"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hiroki","family":"TANAKA","sequence":"additional","affiliation":[{"name":"Nara Institute of Science and Technology"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Satoshi","family":"NAKAMURA","sequence":"additional","affiliation":[{"name":"Nara Institute of Science and Technology"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"532","reference":[{"key":"1","doi-asserted-by":"publisher","unstructured":"[1] G. Skantze, \u201cTurn-taking in conversational systems and human-robot interaction: a review,\u201d Computer Speech &amp; Language, vol.67, p.101178, 2021. 10.1016\/j.csl.2020.101178","DOI":"10.1016\/j.csl.2020.101178"},{"key":"2","doi-asserted-by":"publisher","unstructured":"[2] S.C. Levinson and F. Torreira, \u201cTiming in turn-taking and its implications for processing models of language,\u201d Frontiers in psychology, vol.6, p.731, 2015. 10.3389\/fpsyg.2015.00731","DOI":"10.3389\/fpsyg.2015.00731"},{"key":"3","unstructured":"[3] H.H. Clark, Using language, Cambridge university press, 1996."},{"key":"4","doi-asserted-by":"crossref","unstructured":"[4] T. Stivers, N.J. Enfield, P. Brown, C. Englert, M. Hayashi, T. Heinemann, G. Hoymann, F. Rossano, J.P. De Ruiter, K.E. Yoon, and S.C. Levinson, \u201cUniversals and cultural variation in turn-taking in conversation,\u201d Proc. National Academy of Sciences, vol.106, no.26, pp.10587-10592, 2009. 10.1073\/pnas.0903616106","DOI":"10.1073\/pnas.0903616106"},{"key":"5","doi-asserted-by":"publisher","unstructured":"[5] P.M. Clancy, S.A. Thompson, R. Suzuki, and H. Tao, \u201cThe conversational use of reactive tokens in english, japanese, and mandarin,\u201d Journal of pragmatics, vol.26, no.3, pp.355-387, 1996. 10.1016\/0378-2166(95)00036-4","DOI":"10.1016\/0378-2166(95)00036-4"},{"key":"6","doi-asserted-by":"crossref","unstructured":"[6] F.D. Erickson, \u201cConversational organization: Interaction between speakers and hearers,\u201d 1984.","DOI":"10.1525\/aa.1984.86.3.02a00580"},{"key":"7","doi-asserted-by":"crossref","unstructured":"[7] C. Oertel, M. W\u0142odarczak, J. Edlund, P. Wagner, and J. Gustafson, \u201cGaze patterns in turn-taking,\u201d Thirteenth annual conference of the international speech communication association, 2012. 10.21437\/interspeech.2012-132","DOI":"10.21437\/Interspeech.2012-132"},{"key":"8","doi-asserted-by":"crossref","unstructured":"[8] K. Jokinen, M. Nishida, and S. Yamamoto, \u201cOn eye-gaze and turn-taking,\u201d Proc. 2010 workshop on eye gaze in intelligent human machine interaction, pp.118-123, 2010. 10.1145\/2002333.2002352","DOI":"10.1145\/2002333.2002352"},{"key":"9","doi-asserted-by":"publisher","unstructured":"[9] F. Cummins, \u201cGaze and blinking in dyadic conversation: A study in coordinated behaviour among individuals,\u201d Language and Cognitive Processes, vol.27, no.10, pp.1525-1549, 2012. 10.1080\/01690965.2011.615220","DOI":"10.1080\/01690965.2011.615220"},{"key":"10","doi-asserted-by":"crossref","unstructured":"[10] K. Hara, K. Inoue, K. Takanashi, and T. Kawahara, \u201cTurn-taking prediction based on detection of transition relevance place.,\u201d INTERSPEECH, pp.4170-4174, 2019. 10.21437\/interspeech.2019-1537","DOI":"10.21437\/Interspeech.2019-1537"},{"key":"11","doi-asserted-by":"crossref","unstructured":"[11] B. Inden, Z. Malisz, P. Wagner, and I. Wachsmuth, \u201cTiming and entrainment of multimodal backchanneling behavior for an embodied conversational agent,\u201d Proc. 15th ACM on International conference on multimodal interaction, pp.181-188, 2013. 10.1145\/2522848.2522890","DOI":"10.1145\/2522848.2522890"},{"key":"12","doi-asserted-by":"crossref","unstructured":"[12] B.B. T\u00fcrker, Z. Bu\u00e7inca, E. Erzin, Y. Yemez, and T.M. Sezgin, \u201cAnalysis of engagement and user experience with a laughter responsive social robot.,\u201d Interspeech, pp.844-848, 2017. 10.21437\/interspeech.2017-1395","DOI":"10.21437\/Interspeech.2017-1395"},{"key":"13","doi-asserted-by":"crossref","unstructured":"[13] T. Itoh, N. Kitaoka, and R. Nishimura, \u201cSubjective experiments on influence of response timing in spoken dialogues,\u201d Interspeech, 2009. 10.21437\/interspeech.2009-534","DOI":"10.21437\/Interspeech.2009-534"},{"key":"14","doi-asserted-by":"crossref","unstructured":"[14] N.G. Ward, A.G. Rivera, K. Ward, and D.G. Novick, \u201cRoot causes of lost time and user stress in a simple dialog system,\u201d Interspeech, 2005. 10.21437\/interspeech.2005-458","DOI":"10.21437\/Interspeech.2005-458"},{"key":"15","doi-asserted-by":"crossref","unstructured":"[15] A. Raux, D. Bohus, B. Langner, A.W. Black, and M. Esk\u00e9nazi, \u201cDoing research on a deployed spoken dialogue system: one year of let\u2019s go! experience,\u201d Interspeech, 2006. 10.21437\/interspeech.2006-17","DOI":"10.21437\/Interspeech.2006-17"},{"key":"16","doi-asserted-by":"crossref","unstructured":"[16] S. Fujie, H. Katayama, J. Sakuma, and T. Kobayashi, \u201cTiming generating networks: Neural network based precise turn-taking timing prediction in multiparty conversation,\u201d 22nd Annual Conference of the International Speech Communication Association, INTERSPEECH 2021, pp.3771-3775, International Speech Communication Association, 2021. 10.21437\/interspeech.2021-874","DOI":"10.21437\/Interspeech.2021-874"},{"key":"17","doi-asserted-by":"crossref","unstructured":"[17] J. Sakuma, S. Fujie, and T. Kobayashi, \u201cResponse Timing Estimation for Spoken Dialog System using Dialog Act Estimation,\u201d Proc. Interspeech 2022, pp.4486-4490, 2022. 10.21437\/interspeech.2022-746","DOI":"10.21437\/Interspeech.2022-746"},{"key":"18","doi-asserted-by":"crossref","unstructured":"[18] N.G. Ward, D. Aguirre, G. Cervantes, and O. Fuentes, \u201cTurn-taking predictions across languages and genres using an lstm recurrent neural network,\u201d 2018 IEEE Spoken Language Technology Workshop (SLT), pp.831-837, IEEE, 2018. 10.1109\/slt.2018.8639673","DOI":"10.1109\/SLT.2018.8639673"},{"key":"19","doi-asserted-by":"crossref","unstructured":"[19] D. Lala, K. Inoue, and T. Kawahara, \u201cSmooth turn-taking by a robot using an online continuous model to generate turn-taking cues,\u201d 2019 International Conference on Multimodal Interaction, pp.226-234, 2019. 10.1145\/3340555.3353727","DOI":"10.1145\/3340555.3353727"},{"key":"20","doi-asserted-by":"publisher","unstructured":"[20] K.H. Kendrick, J. Holler, and S.C. Levinson, \u201cTurn-taking in human face-to-face interaction is multimodal: gaze direction and manual gestures aid the coordination of turn transitions,\u201d Philosophical Transactions of the Royal Society B, vol.378, no.1875, p.20210473, 2023. 10.1098\/rstb.2021.0473","DOI":"10.1098\/rstb.2021.0473"},{"key":"21","doi-asserted-by":"crossref","unstructured":"[21] T. Meshorer and P.A. Heeman, \u201cUsing past speaker behavior to better predict turn transitions,\u201d Interspeech, 2016. 10.21437\/interspeech.2016-1409","DOI":"10.21437\/Interspeech.2016-1409"},{"key":"22","doi-asserted-by":"publisher","unstructured":"[22] R. Ishii, X. Ren, M. Muszynski, and L.-P. Morency, \u201cTrimodal prediction of speaking and listening willingness to help improve turn-changing modeling,\u201d Frontiers in Psychology, vol.13, p.774547, 2022. 10.3389\/fpsyg.2022.774547","DOI":"10.3389\/fpsyg.2022.774547"},{"key":"23","doi-asserted-by":"crossref","unstructured":"[23] R. Ishii, X. Ren, M. Muszynski, and L.-P. Morency, \u201cMultimodal and multitask approach to listener\u2019s backchannel prediction: Can prediction of turn-changing and turn-management willingness improve backchannel modeling?,\u201d Proc. 21st ACM International Conference on Intelligent Virtual Agents, pp.131-138, 2021. 10.1145\/3472306.3478360","DOI":"10.1145\/3472306.3478360"},{"key":"24","doi-asserted-by":"crossref","unstructured":"[24] G. Skantze, \u201cTowards a general, continuous model of turn-taking in spoken dialogue using lstm recurrent neural networks,\u201d Proc. 18th Annual SIGdial Meeting on Discourse and Dialogue, pp.220-230, 2017. 10.18653\/v1\/w17-5527","DOI":"10.18653\/v1\/W17-5527"},{"key":"25","doi-asserted-by":"crossref","unstructured":"[25] E. Ekstedt and G. Skantze, \u201cVoice activity projection: Self-supervised learning of turn-taking events,\u201d arXiv preprint arXiv:2205.09812, 2022.","DOI":"10.21437\/Interspeech.2022-10955"},{"key":"26","doi-asserted-by":"crossref","unstructured":"[26] M. Roddy, G. Skantze, and N. Harte, \u201cInvestigating speech features for continuous turn-taking prediction using lstms,\u201d arXiv preprint arXiv:1806.11461, 2018.","DOI":"10.1145\/3242969.3242997"},{"key":"27","doi-asserted-by":"crossref","unstructured":"[27] W. Liermann, Y.H. Park, Y.S. Choi, and K. Lee, \u201cDialogue act-aided backchannel prediction using multi-task learning,\u201d Findings of the Association for Computational Linguistics: EMNLP 2023, pp.15073-15079, 2023. 10.18653\/v1\/2023.findings-emnlp.1006","DOI":"10.18653\/v1\/2023.findings-emnlp.1006"},{"key":"28","doi-asserted-by":"crossref","unstructured":"[28] M. Roddy, G. Skantze, and N. Harte, \u201cMultimodal continuous turn-taking prediction using multiscale rnns,\u201d Proc. 20th ACM International Conference on Multimodal Interaction, pp.186-190, 2018. 10.1145\/3242969.3242997","DOI":"10.1145\/3242969.3242997"},{"key":"29","doi-asserted-by":"crossref","unstructured":"[29] E. Ekstedt and G. Skantze, \u201cHow much does prosody help turn-taking? investigations using voice activity projection models,\u201d arXiv preprint arXiv:2209.05161, 2022.","DOI":"10.18653\/v1\/2022.sigdial-1.51"},{"key":"30","unstructured":"[30] K. Mitsui, Y. Hono, and K. Sawada, \u201cTowards human-like spoken dialogue generation between ai agents from written dialogue,\u201d arXiv preprint arXiv:2310.01088, 2023."},{"key":"31","doi-asserted-by":"crossref","unstructured":"[31] K. Onishi, H. Tanaka, and S. Nakamura, \u201cMultimodal voice activity prediction: Turn-taking events detection in expert-novice conversation,\u201d Proc. 11th Int. Conf. Human-Agent Interaction, pp.13-21, 2023. 10.1145\/3623809.3623837","DOI":"10.1145\/3623809.3623837"},{"key":"32","unstructured":"[32] E. Ekstedt, \u201cVap: Voice activity projection,\u201d 2022.https:\/\/github.com\/ErikEkstedt\/vap_turn_taking"},{"key":"33","doi-asserted-by":"crossref","unstructured":"[33] J.J. Godfrey, E.C. Holliman, and J. McDaniel, \u201cSwitchboard: Telephone speech corpus for research and development,\u201d Acoustics, speech, and signal processing, ieee international conference on, pp.517-520, IEEE Computer Society, 1992. 10.1109\/icassp.1992.225858","DOI":"10.1109\/ICASSP.1992.225858"},{"key":"34","doi-asserted-by":"crossref","unstructured":"[34] A. Cafaro, J. Wagner, T. Baur, S. Dermouche, M. Torres Torres, C. Pelachaud, E. Andr\u00e9, and M. Valstar, \u201cThe noxi database: multimodal recordings of mediated novice-expert interactions,\u201d Proc. 19th ACM Int. Conf. Multimodal Interaction, pp.350-359, 2017. 10.1145\/3136755.3136780","DOI":"10.1145\/3136755.3136780"},{"key":"35","unstructured":"[35] K. Yoshino, H. Tanaka, K. Sugiyama, M. Kondo, and S. Nakamura, \u201cJapanese dialogue corpus of information navigation and attentive listening annotated with extended iso-24617-2 dialogue act tags,\u201d Proc. Eleventh Int. Conf. Language Resources and Evaluation (LREC 2018), 2018."},{"key":"36","unstructured":"[36] A.v.d. Oord, Y. Li, and O. Vinyals, \u201cRepresentation learning with contrastive predictive coding,\u201d arXiv preprint arXiv:1807.03748, 2018."},{"key":"37","doi-asserted-by":"crossref","unstructured":"[37] T. Baltrusaitis, A. Zadeh, Y.C. Lim, and L.P. Morency, \u201cOpenface 2.0: Facial behavior analysis toolkit,\u201d 2018 13th IEEE international conference on automatic face &amp; gesture recognition (FG 2018), pp.59-66, IEEE, 2018. 10.1109\/fg.2018.00019","DOI":"10.1109\/FG.2018.00019"},{"key":"38","doi-asserted-by":"crossref","unstructured":"[38] Z. Cao, G. Hidalgo, T. Simon, S.E. Wei, and Y. Sheikh, \u201cOpenpose: realtime multi-person 2d pose estimation using part affinity fields,\u201d IEEE transactions on pattern analysis and machine intelligence, vol.43, no.1, pp.172-186, 2021. 10.1109\/tpami.2019.2929257","DOI":"10.1109\/TPAMI.2019.2929257"},{"key":"39","doi-asserted-by":"publisher","unstructured":"[39] S. Duncan, \u201cSome signals and rules for taking speaking turns in conversations.,\u201d Journal of personality and social psychology, vol.23, no.2, p.283, 1972. 10.1037\/h0033031","DOI":"10.1037\/h0033031"},{"key":"40","doi-asserted-by":"crossref","unstructured":"[40] M. Zellers, D. House, and S. Alexanderson, \u201cProsody and hand gesture at turn boundaries in swedish,\u201d Proc. Speech Prosody 2016, pp.831-835, 2016. 10.21437\/speechprosody.2016-170","DOI":"10.21437\/SpeechProsody.2016-170"},{"key":"41","doi-asserted-by":"publisher","unstructured":"[41] J. Holler, K.H. Kendrick, and S.C. Levinson, \u201cProcessing language in face-to-face conversation: Questions with gestures get faster responses,\u201d Psychonomic bulletin &amp; review, vol.25, pp.1900-1908, 2018. 10.3758\/s13423-017-1363-z","DOI":"10.3758\/s13423-017-1363-z"},{"key":"42","doi-asserted-by":"crossref","unstructured":"[42] J. Streeck and U. Hartge, \u201cPreviews: Gestures at the transition place,\u201d The contextualization of language, pp.135-157, 1992. 10.1075\/pbns.22.10str","DOI":"10.1075\/pbns.22.10str"},{"key":"43","unstructured":"[43] A. Paszke, S. Gross, F. Massa, A. Lerer, J. Bradbury, G. Chanan, T. Killeen, Z. Lin, N. Gimelshein, L. Antiga, et al., \u201cPytorch: An imperative style, high-performance deep learning library,\u201d Advances in neural information processing systems, vol.32, 2019."},{"key":"44","doi-asserted-by":"crossref","unstructured":"[44] K. Saleh, K. Yu, and F. Chen, \u201cImproving users engagement detection using end-to-end spatio-temporal convolutional neural networks,\u201d Companion of the 2021 ACM\/IEEE Int. Conf. Human-Robot Interaction, pp.190-194, 2021. 10.1145\/3434074.3447157","DOI":"10.1145\/3434074.3447157"}],"container-title":["IEICE Transactions on Information and Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E108.D\/6\/E108.D_2024HCP0002\/_pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,7]],"date-time":"2025-06-07T03:42:33Z","timestamp":1749267753000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E108.D\/6\/E108.D_2024HCP0002\/_article"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,1]]},"references-count":44,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2025]]}},"URL":"https:\/\/doi.org\/10.1587\/transinf.2024hcp0002","relation":{},"ISSN":["0916-8532","1745-1361"],"issn-type":[{"value":"0916-8532","type":"print"},{"value":"1745-1361","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,6,1]]},"article-number":"2024HCP0002"}}